@motionscript/molecule 0.0.0-stage → 0.1.0-alpha.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (53) hide show
  1. package/CHANGELOG.md +5 -0
  2. package/LICENSE +201 -0
  3. package/dist/browser/index.js +4 -0
  4. package/dist/browser/index.js.map +7 -0
  5. package/dist/browser/manifest.json +11 -0
  6. package/dist/index.d.ts +3 -0
  7. package/dist/index.d.ts.map +1 -0
  8. package/dist/index.js +3 -0
  9. package/dist/index.js.map +1 -0
  10. package/dist/nodes.d.ts +18 -0
  11. package/dist/nodes.d.ts.map +1 -0
  12. package/dist/nodes.js +18 -0
  13. package/dist/nodes.js.map +1 -0
  14. package/dist/protein/chemistry.d.ts +52 -0
  15. package/dist/protein/chemistry.d.ts.map +1 -0
  16. package/dist/protein/chemistry.js +208 -0
  17. package/dist/protein/chemistry.js.map +1 -0
  18. package/dist/protein/index.d.ts +35 -0
  19. package/dist/protein/index.d.ts.map +1 -0
  20. package/dist/protein/index.js +35 -0
  21. package/dist/protein/index.js.map +1 -0
  22. package/dist/protein/parse.d.ts +32 -0
  23. package/dist/protein/parse.d.ts.map +1 -0
  24. package/dist/protein/parse.js +387 -0
  25. package/dist/protein/parse.js.map +1 -0
  26. package/dist/protein/protein.d.ts +265 -0
  27. package/dist/protein/protein.d.ts.map +1 -0
  28. package/dist/protein/protein.js +645 -0
  29. package/dist/protein/protein.js.map +1 -0
  30. package/dist/protein/ribbon.d.ts +83 -0
  31. package/dist/protein/ribbon.d.ts.map +1 -0
  32. package/dist/protein/ribbon.js +468 -0
  33. package/dist/protein/ribbon.js.map +1 -0
  34. package/dist/protein/shared.d.ts +221 -0
  35. package/dist/protein/shared.d.ts.map +1 -0
  36. package/dist/protein/shared.js +478 -0
  37. package/dist/protein/shared.js.map +1 -0
  38. package/dist/protein/structure.d.ts +184 -0
  39. package/dist/protein/structure.d.ts.map +1 -0
  40. package/dist/protein/structure.js +324 -0
  41. package/dist/protein/structure.js.map +1 -0
  42. package/package.json +64 -3
  43. package/registry.json +22 -0
  44. package/src/index.ts +2 -0
  45. package/src/nodes.ts +18 -0
  46. package/src/protein/chemistry.ts +223 -0
  47. package/src/protein/index.ts +34 -0
  48. package/src/protein/parse.ts +427 -0
  49. package/src/protein/protein.ts +897 -0
  50. package/src/protein/ribbon.ts +658 -0
  51. package/src/protein/shared.ts +622 -0
  52. package/src/protein/structure.ts +491 -0
  53. package/README.md +0 -4
@@ -0,0 +1,387 @@
1
+ /**
2
+ * Reads a coordinate file into a {@link ProteinStructure}.
3
+ *
4
+ * Two formats, because the Protein Data Bank serves two and which one an entry
5
+ * has is not the author's choice: the legacy **PDB** format numbers atoms in
6
+ * five columns and names chains in one, so an entry that outgrew either — a
7
+ * ribosome, a capsid — exists only as **mmCIF**. Refusing the second would mean
8
+ * refusing exactly the structures most worth looking at.
9
+ *
10
+ * Both are read the same way: pull out an atom list and whatever secondary
11
+ * structure the file annotates, then hand both to `deriveStructure`, which does
12
+ * the geometry. {@link parseStructure} sniffs which is which, so a caller
13
+ * holding a downloaded file never has to.
14
+ *
15
+ * **The first model only.** An NMR ensemble is twenty superposed conformations
16
+ * of the same molecule; drawing all of them at once produces a blur, and
17
+ * choosing between them is a question this node does not ask. The first is the
18
+ * conventional representative.
19
+ */
20
+ import { elementOf, residueKind } from "./chemistry.js";
21
+ import { deriveStructure, EMPTY_STRUCTURE, } from "./structure.js";
22
+ /**
23
+ * Reads a coordinate file, detecting its format from the contents.
24
+ *
25
+ * `id` labels the result — an accession, or the asset's name. Anything that
26
+ * can't be read at all comes back as {@link EMPTY_STRUCTURE} rather than
27
+ * throwing: this runs against a file someone just uploaded or an accession
28
+ * someone is halfway through typing, and both of those are ordinary states
29
+ * rather than errors. An empty structure draws nothing, which is the honest
30
+ * picture of a file with no atoms in it.
31
+ */
32
+ export function parseStructure(text, id = "") {
33
+ if (typeof text !== "string" || text.trim() === "") {
34
+ return { ...EMPTY_STRUCTURE, id };
35
+ }
36
+ return isMmcif(text) ? parseMmcif(text, id) : parsePdb(text, id);
37
+ }
38
+ /**
39
+ * Whether a file is mmCIF.
40
+ *
41
+ * The `data_` block header is mmCIF's first line and appears in no PDB file, so
42
+ * it decides on its own — but only over the first stretch, since `data_` is also
43
+ * an ordinary substring that could turn up in a PDB `REMARK`. The category
44
+ * prefix is the belt-and-braces check for a fragment handed over without its
45
+ * header.
46
+ */
47
+ function isMmcif(text) {
48
+ const head = text.slice(0, 4096);
49
+ return /^\s*data_/.test(head) || head.includes("_atom_site.");
50
+ }
51
+ // --- PDB -------------------------------------------------------------------
52
+ /**
53
+ * Reads the legacy PDB format, which is **fixed-column**: every field is at a
54
+ * known offset and the whitespace between them is padding rather than a
55
+ * separator. Splitting on spaces is the classic way to get this wrong — a
56
+ * residue number that runs into its insertion code, or a `-` sign that eats the
57
+ * gap between two coordinates, and the record silently comes apart.
58
+ *
59
+ * Columns, 1-based as the specification numbers them:
60
+ *
61
+ * 7–11 serial 13–16 name 18–20 resName 22 chainID
62
+ * 23–26 resSeq 31–38 x 39–46 y 47–54 z
63
+ * 77–78 element
64
+ */
65
+ function parsePdb(text, id) {
66
+ const atoms = [];
67
+ const spans = [];
68
+ let title = "";
69
+ for (const line of text.split("\n")) {
70
+ const record = line.slice(0, 6).trim();
71
+ // Everything after the first `ENDMDL` is another conformation of what we
72
+ // already have. Stopping at it rather than filtering by model number also
73
+ // ends the read early on a large ensemble.
74
+ if (record === "ENDMDL")
75
+ break;
76
+ if (record === "ATOM" || record === "HETATM") {
77
+ const atom = parsePdbAtom(line, record === "HETATM");
78
+ if (atom)
79
+ atoms.push(atom);
80
+ continue;
81
+ }
82
+ if (record === "TITLE") {
83
+ // A long title is continued across several records, with the continuation
84
+ // number in columns 9–10 and the text always from column 11.
85
+ title = `${title} ${line.slice(10).trim()}`.trim();
86
+ continue;
87
+ }
88
+ // The two annotation records. Their first-residue fields sit at *different*
89
+ // offsets from each other — a genuine wart of the format rather than a
90
+ // mistake here — while their last-residue fields agree.
91
+ if (record === "HELIX") {
92
+ const span = pdbSpan(line, 19, 21, "helix");
93
+ if (span)
94
+ spans.push(span);
95
+ continue;
96
+ }
97
+ if (record === "SHEET") {
98
+ const span = pdbSpan(line, 21, 22, "sheet");
99
+ if (span)
100
+ spans.push(span);
101
+ }
102
+ }
103
+ return deriveStructure(id, title, atoms, spans);
104
+ }
105
+ /** One `ATOM`/`HETATM` record, or `null` when its coordinates don't read. */
106
+ function parsePdbAtom(line, hetero) {
107
+ // Truncated before the coordinates. Worth checking outright: a short slice of
108
+ // a fixed-column record reads as the empty string, and `Number("")` is 0
109
+ // rather than `NaN` — so without this a mangled line becomes an atom sitting
110
+ // at the origin, which is far harder to notice than a missing one.
111
+ if (line.length < 54)
112
+ return null;
113
+ // An alternate location: the same atom modelled twice because the side chain
114
+ // is disordered. Keeping both would double the sticks through that residue,
115
+ // so take the first (blank or `A`), which is the convention.
116
+ const altLoc = line.slice(16, 17).trim();
117
+ if (altLoc !== "" && altLoc !== "A")
118
+ return null;
119
+ const x = Number(line.slice(30, 38));
120
+ const y = Number(line.slice(38, 46));
121
+ const z = Number(line.slice(46, 54));
122
+ if (!Number.isFinite(x) || !Number.isFinite(y) || !Number.isFinite(z)) {
123
+ return null;
124
+ }
125
+ const name = line.slice(12, 16).trim();
126
+ const residue = line.slice(17, 20).trim();
127
+ const kind = residueKind(residue);
128
+ return {
129
+ serial: Number(line.slice(6, 11)) || 0,
130
+ name,
131
+ element: elementOf(line.slice(76, 78), name, kind),
132
+ residue,
133
+ residueSeq: Number(line.slice(22, 26)) || 0,
134
+ // A file with a single unnamed chain leaves the column blank; calling that
135
+ // `A` keeps "group by chain" from producing one group named nothing.
136
+ chain: line.slice(21, 22).trim() || "A",
137
+ x,
138
+ y,
139
+ z,
140
+ hetero,
141
+ kind,
142
+ };
143
+ }
144
+ /** Where the last residue's number sits on both `HELIX` and `SHEET` — columns 34–37. */
145
+ const PDB_SPAN_END_AT = 33;
146
+ /**
147
+ * One `HELIX`/`SHEET` record as a residue span.
148
+ *
149
+ * The first-residue offsets differ between the two records, so they are passed
150
+ * in rather than hard-coded: a helix names its chain in column 20 and its first
151
+ * residue in 22–25, a sheet names its chain in 22 and its first residue in
152
+ * 23–26. Their *last* residue is in the same place in both, which is the one
153
+ * thing they agree on.
154
+ */
155
+ function pdbSpan(line, chainAt, startAt, kind) {
156
+ const chain = line.slice(chainAt, chainAt + 1).trim() || "A";
157
+ // Read as text first: a blank fixed-column field slices to `""`, and
158
+ // `Number("")` is 0, so a `Number.isFinite` check alone would accept a record
159
+ // that states nothing as a span over residue 0.
160
+ const startText = line.slice(startAt, startAt + 4).trim();
161
+ const endText = line.slice(PDB_SPAN_END_AT, PDB_SPAN_END_AT + 4).trim();
162
+ if (startText === "" || endText === "")
163
+ return null;
164
+ const start = Number(startText);
165
+ const end = Number(endText);
166
+ if (!Number.isFinite(start) || !Number.isFinite(end))
167
+ return null;
168
+ return { chain, start, end, kind };
169
+ }
170
+ // --- mmCIF -----------------------------------------------------------------
171
+ /**
172
+ * Reads mmCIF, which is a tagged format rather than a positional one: values are
173
+ * whitespace-separated and their meaning comes from the column headers declared
174
+ * above them, so the reader has to learn the layout before it can read a row.
175
+ *
176
+ * Only three categories are consulted — the atoms, and the two that annotate
177
+ * secondary structure. mmCIF carries dozens more, and every one this doesn't
178
+ * read is one this can't be broken by.
179
+ */
180
+ function parseMmcif(text, id) {
181
+ const lines = text.split("\n");
182
+ const atoms = mmcifAtoms(lines);
183
+ const spans = [
184
+ ...mmcifSpans(lines, "_struct_conf", "helix"),
185
+ ...mmcifSpans(lines, "_struct_sheet_range", "sheet"),
186
+ ];
187
+ return deriveStructure(id, mmcifTitle(lines), atoms, spans);
188
+ }
189
+ /** The atoms of the first model in an `_atom_site` loop. */
190
+ function mmcifAtoms(lines) {
191
+ const atoms = [];
192
+ let model = null;
193
+ for (const row of mmcifRows(lines, "_atom_site")) {
194
+ // `auth_*` is the numbering the literature uses and the one a PDB file would
195
+ // have carried; `label_*` is the internal, re-derived scheme. Preferring
196
+ // auth means a residue number quoted in a paper matches what is drawn, and
197
+ // falling back keeps a file that omits it readable.
198
+ const chain = row("auth_asym_id") || row("label_asym_id") || "A";
199
+ const residue = row("auth_comp_id") || row("label_comp_id");
200
+ const name = row("auth_atom_id") || row("label_atom_id");
201
+ const thisModel = row("pdbx_PDB_model_num");
202
+ if (model === null)
203
+ model = thisModel;
204
+ else if (thisModel !== model)
205
+ break;
206
+ const altLoc = row("label_alt_id");
207
+ if (altLoc !== "" && altLoc !== "." && altLoc !== "?" && altLoc !== "A") {
208
+ continue;
209
+ }
210
+ const x = Number(row("Cartn_x"));
211
+ const y = Number(row("Cartn_y"));
212
+ const z = Number(row("Cartn_z"));
213
+ if (!Number.isFinite(x) || !Number.isFinite(y) || !Number.isFinite(z)) {
214
+ continue;
215
+ }
216
+ const kind = residueKind(residue);
217
+ atoms.push({
218
+ serial: Number(row("id")) || 0,
219
+ name,
220
+ element: elementOf(row("type_symbol"), name, kind),
221
+ residue,
222
+ residueSeq: Number(row("auth_seq_id") || row("label_seq_id")) || 0,
223
+ chain,
224
+ x,
225
+ y,
226
+ z,
227
+ hetero: row("group_PDB") === "HETATM",
228
+ kind,
229
+ });
230
+ }
231
+ return atoms;
232
+ }
233
+ /** The residue spans of one annotation category. */
234
+ function mmcifSpans(lines, category, kind) {
235
+ const spans = [];
236
+ for (const row of mmcifRows(lines, category)) {
237
+ // `_struct_conf` covers turns and bends as well as helices, all under one
238
+ // category and told apart by this tag. A sheet range has no such tag, so an
239
+ // absent one passes.
240
+ const type = row("conf_type_id");
241
+ if (type !== "" && !type.startsWith("HELX"))
242
+ continue;
243
+ const chain = row("beg_auth_asym_id") || row("beg_label_asym_id") || "A";
244
+ const start = Number(row("beg_auth_seq_id") || row("beg_label_seq_id"));
245
+ const end = Number(row("end_auth_seq_id") || row("end_label_seq_id"));
246
+ if (!Number.isFinite(start) || !Number.isFinite(end))
247
+ continue;
248
+ spans.push({ chain, start, end, kind });
249
+ }
250
+ return spans;
251
+ }
252
+ /** The entry's title, from the `_struct.title` item. */
253
+ function mmcifTitle(lines) {
254
+ for (let i = 0; i < lines.length; i++) {
255
+ const line = lines[i];
256
+ if (!line.startsWith("_struct.title"))
257
+ continue;
258
+ const inline = line.slice("_struct.title".length).trim();
259
+ // A value too long for the line is carried on the next one, or in a
260
+ // semicolon-delimited block below it; the inline form covers the rest.
261
+ if (inline !== "")
262
+ return unquote(inline);
263
+ const next = lines[i + 1]?.trim() ?? "";
264
+ return unquote(next.startsWith(";") ? next.slice(1) : next);
265
+ }
266
+ return "";
267
+ }
268
+ /**
269
+ * Walks the rows of one mmCIF category, yielding a field reader for each.
270
+ *
271
+ * The reader is a closure over the current row rather than an object, so
272
+ * nothing allocates a record per atom — this runs over hundreds of thousands of
273
+ * rows on a large entry, and the caller only ever wants a handful of the fields.
274
+ *
275
+ * Handles both forms a category takes: a `loop_` with a header block and many
276
+ * rows, and the flat `_category.field value` form mmCIF uses when there is
277
+ * exactly one row.
278
+ */
279
+ function* mmcifRows(lines, category) {
280
+ const prefix = `${category}.`;
281
+ for (let i = 0; i < lines.length; i++) {
282
+ const line = lines[i].trim();
283
+ // The flat, single-row form: consecutive `_category.field value` lines.
284
+ if (line.startsWith(prefix)) {
285
+ const single = new Map();
286
+ while (i < lines.length) {
287
+ const item = lines[i].trim();
288
+ if (!item.startsWith(prefix))
289
+ break;
290
+ const gap = item.search(/\s/);
291
+ if (gap === -1) {
292
+ // A field whose value is on the following line.
293
+ single.set(item.slice(prefix.length), unquote(lines[++i]?.trim() ?? ""));
294
+ }
295
+ else {
296
+ single.set(item.slice(prefix.length, gap), unquote(item.slice(gap).trim()));
297
+ }
298
+ i++;
299
+ }
300
+ yield (field) => single.get(field) ?? "";
301
+ return;
302
+ }
303
+ if (line !== "loop_")
304
+ continue;
305
+ // The header block: every `_category.field` line, in the order the values
306
+ // will arrive in.
307
+ const columns = new Map();
308
+ let cursor = i + 1;
309
+ while (cursor < lines.length) {
310
+ const header = lines[cursor].trim();
311
+ if (!header.startsWith("_"))
312
+ break;
313
+ if (header.startsWith(prefix)) {
314
+ columns.set(header.slice(prefix.length), columns.size);
315
+ }
316
+ cursor++;
317
+ }
318
+ // A `loop_` for some other category. Its rows carry no leading underscore
319
+ // and aren't `loop_`, so the outer scan walks past them harmlessly.
320
+ if (columns.size === 0)
321
+ continue;
322
+ for (; cursor < lines.length; cursor++) {
323
+ const row = lines[cursor];
324
+ const trimmed = row.trim();
325
+ // A loop ends at the next block, the next loop, or a blank line.
326
+ if (trimmed === "" || trimmed === "#" || trimmed.startsWith("_"))
327
+ break;
328
+ if (trimmed === "loop_" || trimmed.startsWith("data_"))
329
+ break;
330
+ const values = splitCifRow(trimmed);
331
+ yield (field) => {
332
+ const index = columns.get(field);
333
+ if (index === undefined)
334
+ return "";
335
+ const value = values[index] ?? "";
336
+ return value === "." || value === "?" ? "" : value;
337
+ };
338
+ }
339
+ return;
340
+ }
341
+ }
342
+ /**
343
+ * Splits one mmCIF row into values, respecting quotes.
344
+ *
345
+ * Quoting is not decoration here: a chain identifier can be `'A'`, and a residue
346
+ * name can contain a space that a plain `split` would turn into two columns and
347
+ * shift every field after it by one.
348
+ */
349
+ function splitCifRow(row) {
350
+ const values = [];
351
+ let i = 0;
352
+ while (i < row.length) {
353
+ const char = row[i];
354
+ if (char === " " || char === "\t") {
355
+ i++;
356
+ continue;
357
+ }
358
+ if (char === "'" || char === '"') {
359
+ const end = row.indexOf(char, i + 1);
360
+ if (end === -1) {
361
+ values.push(row.slice(i + 1));
362
+ break;
363
+ }
364
+ values.push(row.slice(i + 1, end));
365
+ i = end + 1;
366
+ continue;
367
+ }
368
+ let end = i;
369
+ while (end < row.length && row[end] !== " " && row[end] !== "\t")
370
+ end++;
371
+ values.push(row.slice(i, end));
372
+ i = end;
373
+ }
374
+ return values;
375
+ }
376
+ /** Strips the quotes mmCIF wraps a value in, and the null placeholders. */
377
+ function unquote(value) {
378
+ const text = value.trim();
379
+ if (text === "." || text === "?")
380
+ return "";
381
+ const first = text[0];
382
+ if ((first === "'" || first === '"') && text.endsWith(first) && text.length > 1) {
383
+ return text.slice(1, -1);
384
+ }
385
+ return text;
386
+ }
387
+ //# sourceMappingURL=parse.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"parse.js","sourceRoot":"","sources":["../../src/protein/parse.ts"],"names":[],"mappings":"AAAA;;;;;;;;;;;;;;;;;;GAkBG;AAEH,OAAO,EAAE,SAAS,EAAE,WAAW,EAAE,MAAM,aAAa,CAAA;AACpD,OAAO,EACL,eAAe,EACf,eAAe,GAIhB,MAAM,aAAa,CAAA;AAEpB;;;;;;;;;GASG;AACH,MAAM,UAAU,cAAc,CAAC,IAAY,EAAE,EAAE,GAAG,EAAE;IAClD,IAAI,OAAO,IAAI,KAAK,QAAQ,IAAI,IAAI,CAAC,IAAI,EAAE,KAAK,EAAE,EAAE,CAAC;QACnD,OAAO,EAAE,GAAG,eAAe,EAAE,EAAE,EAAE,CAAA;IACnC,CAAC;IACD,OAAO,OAAO,CAAC,IAAI,CAAC,CAAC,CAAC,CAAC,UAAU,CAAC,IAAI,EAAE,EAAE,CAAC,CAAC,CAAC,CAAC,QAAQ,CAAC,IAAI,EAAE,EAAE,CAAC,CAAA;AAClE,CAAC;AAED;;;;;;;;GAQG;AACH,SAAS,OAAO,CAAC,IAAY;IAC3B,MAAM,IAAI,GAAG,IAAI,CAAC,KAAK,CAAC,CAAC,EAAE,IAAI,CAAC,CAAA;IAChC,OAAO,WAAW,CAAC,IAAI,CAAC,IAAI,CAAC,IAAI,IAAI,CAAC,QAAQ,CAAC,aAAa,CAAC,CAAA;AAC/D,CAAC;AAED,8EAA8E;AAE9E;;;;;;;;;;;;GAYG;AACH,SAAS,QAAQ,CAAC,IAAY,EAAE,EAAU;IACxC,MAAM,KAAK,GAAkB,EAAE,CAAA;IAC/B,MAAM,KAAK,GAAoB,EAAE,CAAA;IACjC,IAAI,KAAK,GAAG,EAAE,CAAA;IAEd,KAAK,MAAM,IAAI,IAAI,IAAI,CAAC,KAAK,CAAC,IAAI,CAAC,EAAE,CAAC;QACpC,MAAM,MAAM,GAAG,IAAI,CAAC,KAAK,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC,IAAI,EAAE,CAAA;QAEtC,yEAAyE;QACzE,0EAA0E;QAC1E,2CAA2C;QAC3C,IAAI,MAAM,KAAK,QAAQ;YAAE,MAAK;QAE9B,IAAI,MAAM,KAAK,MAAM,IAAI,MAAM,KAAK,QAAQ,EAAE,CAAC;YAC7C,MAAM,IAAI,GAAG,YAAY,CAAC,IAAI,EAAE,MAAM,KAAK,QAAQ,CAAC,CAAA;YACpD,IAAI,IAAI;gBAAE,KAAK,CAAC,IAAI,CAAC,IAAI,CAAC,CAAA;YAC1B,SAAQ;QACV,CAAC;QAED,IAAI,MAAM,KAAK,OAAO,EAAE,CAAC;YACvB,0EAA0E;YAC1E,6DAA6D;YAC7D,KAAK,GAAG,GAAG,KAAK,IAAI,IAAI,CAAC,KAAK,CAAC,EAAE,CAAC,CAAC,IAAI,EAAE,EAAE,CAAC,IAAI,EAAE,CAAA;YAClD,SAAQ;QACV,CAAC;QAED,4EAA4E;QAC5E,uEAAuE;QACvE,wDAAwD;QACxD,IAAI,MAAM,KAAK,OAAO,EAAE,CAAC;YACvB,MAAM,IAAI,GAAG,OAAO,CAAC,IAAI,EAAE,EAAE,EAAE,EAAE,EAAE,OAAO,CAAC,CAAA;YAC3C,IAAI,IAAI;gBAAE,KAAK,CAAC,IAAI,CAAC,IAAI,CAAC,CAAA;YAC1B,SAAQ;QACV,CAAC;QACD,IAAI,MAAM,KAAK,OAAO,EAAE,CAAC;YACvB,MAAM,IAAI,GAAG,OAAO,CAAC,IAAI,EAAE,EAAE,EAAE,EAAE,EAAE,OAAO,CAAC,CAAA;YAC3C,IAAI,IAAI;gBAAE,KAAK,CAAC,IAAI,CAAC,IAAI,CAAC,CAAA;QAC5B,CAAC;IACH,CAAC;IAED,OAAO,eAAe,CAAC,EAAE,EAAE,KAAK,EAAE,KAAK,EAAE,KAAK,CAAC,CAAA;AACjD,CAAC;AAED,6EAA6E;AAC7E,SAAS,YAAY,CAAC,IAAY,EAAE,MAAe;IACjD,8EAA8E;IAC9E,yEAAyE;IACzE,6EAA6E;IAC7E,mEAAmE;IACnE,IAAI,IAAI,CAAC,MAAM,GAAG,EAAE;QAAE,OAAO,IAAI,CAAA;IAEjC,6EAA6E;IAC7E,4EAA4E;IAC5E,6DAA6D;IAC7D,MAAM,MAAM,GAAG,IAAI,CAAC,KAAK,CAAC,EAAE,EAAE,EAAE,CAAC,CAAC,IAAI,EAAE,CAAA;IACxC,IAAI,MAAM,KAAK,EAAE,IAAI,MAAM,KAAK,GAAG;QAAE,OAAO,IAAI,CAAA;IAEhD,MAAM,CAAC,GAAG,MAAM,CAAC,IAAI,CAAC,KAAK,CAAC,EAAE,EAAE,EAAE,CAAC,CAAC,CAAA;IACpC,MAAM,CAAC,GAAG,MAAM,CAAC,IAAI,CAAC,KAAK,CAAC,EAAE,EAAE,EAAE,CAAC,CAAC,CAAA;IACpC,MAAM,CAAC,GAAG,MAAM,CAAC,IAAI,CAAC,KAAK,CAAC,EAAE,EAAE,EAAE,CAAC,CAAC,CAAA;IACpC,IAAI,CAAC,MAAM,CAAC,QAAQ,CAAC,CAAC,CAAC,IAAI,CAAC,MAAM,CAAC,QAAQ,CAAC,CAAC,CAAC,IAAI,CAAC,MAAM,CAAC,QAAQ,CAAC,CAAC,CAAC,EAAE,CAAC;QACtE,OAAO,IAAI,CAAA;IACb,CAAC;IAED,MAAM,IAAI,GAAG,IAAI,CAAC,KAAK,CAAC,EAAE,EAAE,EAAE,CAAC,CAAC,IAAI,EAAE,CAAA;IACtC,MAAM,OAAO,GAAG,IAAI,CAAC,KAAK,CAAC,EAAE,EAAE,EAAE,CAAC,CAAC,IAAI,EAAE,CAAA;IACzC,MAAM,IAAI,GAAG,WAAW,CAAC,OAAO,CAAC,CAAA;IAEjC,OAAO;QACL,MAAM,EAAE,MAAM,CAAC,IAAI,CAAC,KAAK,CAAC,CAAC,EAAE,EAAE,CAAC,CAAC,IAAI,CAAC;QACtC,IAAI;QACJ,OAAO,EAAE,SAAS,CAAC,IAAI,CAAC,KAAK,CAAC,EAAE,EAAE,EAAE,CAAC,EAAE,IAAI,EAAE,IAAI,CAAC;QAClD,OAAO;QACP,UAAU,EAAE,MAAM,CAAC,IAAI,CAAC,KAAK,CAAC,EAAE,EAAE,EAAE,CAAC,CAAC,IAAI,CAAC;QAC3C,2EAA2E;QAC3E,qEAAqE;QACrE,KAAK,EAAE,IAAI,CAAC,KAAK,CAAC,EAAE,EAAE,EAAE,CAAC,CAAC,IAAI,EAAE,IAAI,GAAG;QACvC,CAAC;QACD,CAAC;QACD,CAAC;QACD,MAAM;QACN,IAAI;KACL,CAAA;AACH,CAAC;AAED,wFAAwF;AACxF,MAAM,eAAe,GAAG,EAAE,CAAA;AAE1B;;;;;;;;GAQG;AACH,SAAS,OAAO,CACd,IAAY,EACZ,OAAe,EACf,OAAe,EACf,IAA2B;IAE3B,MAAM,KAAK,GAAG,IAAI,CAAC,KAAK,CAAC,OAAO,EAAE,OAAO,GAAG,CAAC,CAAC,CAAC,IAAI,EAAE,IAAI,GAAG,CAAA;IAC5D,qEAAqE;IACrE,8EAA8E;IAC9E,gDAAgD;IAChD,MAAM,SAAS,GAAG,IAAI,CAAC,KAAK,CAAC,OAAO,EAAE,OAAO,GAAG,CAAC,CAAC,CAAC,IAAI,EAAE,CAAA;IACzD,MAAM,OAAO,GAAG,IAAI,CAAC,KAAK,CAAC,eAAe,EAAE,eAAe,GAAG,CAAC,CAAC,CAAC,IAAI,EAAE,CAAA;IACvE,IAAI,SAAS,KAAK,EAAE,IAAI,OAAO,KAAK,EAAE;QAAE,OAAO,IAAI,CAAA;IAEnD,MAAM,KAAK,GAAG,MAAM,CAAC,SAAS,CAAC,CAAA;IAC/B,MAAM,GAAG,GAAG,MAAM,CAAC,OAAO,CAAC,CAAA;IAC3B,IAAI,CAAC,MAAM,CAAC,QAAQ,CAAC,KAAK,CAAC,IAAI,CAAC,MAAM,CAAC,QAAQ,CAAC,GAAG,CAAC;QAAE,OAAO,IAAI,CAAA;IACjE,OAAO,EAAE,KAAK,EAAE,KAAK,EAAE,GAAG,EAAE,IAAI,EAAE,CAAA;AACpC,CAAC;AAED,8EAA8E;AAE9E;;;;;;;;GAQG;AACH,SAAS,UAAU,CAAC,IAAY,EAAE,EAAU;IAC1C,MAAM,KAAK,GAAG,IAAI,CAAC,KAAK,CAAC,IAAI,CAAC,CAAA;IAC9B,MAAM,KAAK,GAAG,UAAU,CAAC,KAAK,CAAC,CAAA;IAC/B,MAAM,KAAK,GAAG;QACZ,GAAG,UAAU,CAAC,KAAK,EAAE,cAAc,EAAE,OAAO,CAAC;QAC7C,GAAG,UAAU,CAAC,KAAK,EAAE,qBAAqB,EAAE,OAAO,CAAC;KACrD,CAAA;IACD,OAAO,eAAe,CAAC,EAAE,EAAE,UAAU,CAAC,KAAK,CAAC,EAAE,KAAK,EAAE,KAAK,CAAC,CAAA;AAC7D,CAAC;AAED,4DAA4D;AAC5D,SAAS,UAAU,CAAC,KAAe;IACjC,MAAM,KAAK,GAAkB,EAAE,CAAA;IAC/B,IAAI,KAAK,GAAkB,IAAI,CAAA;IAE/B,KAAK,MAAM,GAAG,IAAI,SAAS,CAAC,KAAK,EAAE,YAAY,CAAC,EAAE,CAAC;QACjD,6EAA6E;QAC7E,yEAAyE;QACzE,2EAA2E;QAC3E,oDAAoD;QACpD,MAAM,KAAK,GAAG,GAAG,CAAC,cAAc,CAAC,IAAI,GAAG,CAAC,eAAe,CAAC,IAAI,GAAG,CAAA;QAChE,MAAM,OAAO,GAAG,GAAG,CAAC,cAAc,CAAC,IAAI,GAAG,CAAC,eAAe,CAAC,CAAA;QAC3D,MAAM,IAAI,GAAG,GAAG,CAAC,cAAc,CAAC,IAAI,GAAG,CAAC,eAAe,CAAC,CAAA;QAExD,MAAM,SAAS,GAAG,GAAG,CAAC,oBAAoB,CAAC,CAAA;QAC3C,IAAI,KAAK,KAAK,IAAI;YAAE,KAAK,GAAG,SAAS,CAAA;aAChC,IAAI,SAAS,KAAK,KAAK;YAAE,MAAK;QAEnC,MAAM,MAAM,GAAG,GAAG,CAAC,cAAc,CAAC,CAAA;QAClC,IAAI,MAAM,KAAK,EAAE,IAAI,MAAM,KAAK,GAAG,IAAI,MAAM,KAAK,GAAG,IAAI,MAAM,KAAK,GAAG,EAAE,CAAC;YACxE,SAAQ;QACV,CAAC;QAED,MAAM,CAAC,GAAG,MAAM,CAAC,GAAG,CAAC,SAAS,CAAC,CAAC,CAAA;QAChC,MAAM,CAAC,GAAG,MAAM,CAAC,GAAG,CAAC,SAAS,CAAC,CAAC,CAAA;QAChC,MAAM,CAAC,GAAG,MAAM,CAAC,GAAG,CAAC,SAAS,CAAC,CAAC,CAAA;QAChC,IAAI,CAAC,MAAM,CAAC,QAAQ,CAAC,CAAC,CAAC,IAAI,CAAC,MAAM,CAAC,QAAQ,CAAC,CAAC,CAAC,IAAI,CAAC,MAAM,CAAC,QAAQ,CAAC,CAAC,CAAC,EAAE,CAAC;YACtE,SAAQ;QACV,CAAC;QAED,MAAM,IAAI,GAAG,WAAW,CAAC,OAAO,CAAC,CAAA;QACjC,KAAK,CAAC,IAAI,CAAC;YACT,MAAM,EAAE,MAAM,CAAC,GAAG,CAAC,IAAI,CAAC,CAAC,IAAI,CAAC;YAC9B,IAAI;YACJ,OAAO,EAAE,SAAS,CAAC,GAAG,CAAC,aAAa,CAAC,EAAE,IAAI,EAAE,IAAI,CAAC;YAClD,OAAO;YACP,UAAU,EAAE,MAAM,CAAC,GAAG,CAAC,aAAa,CAAC,IAAI,GAAG,CAAC,cAAc,CAAC,CAAC,IAAI,CAAC;YAClE,KAAK;YACL,CAAC;YACD,CAAC;YACD,CAAC;YACD,MAAM,EAAE,GAAG,CAAC,WAAW,CAAC,KAAK,QAAQ;YACrC,IAAI;SACL,CAAC,CAAA;IACJ,CAAC;IAED,OAAO,KAAK,CAAA;AACd,CAAC;AAED,oDAAoD;AACpD,SAAS,UAAU,CACjB,KAAe,EACf,QAAgB,EAChB,IAA2B;IAE3B,MAAM,KAAK,GAAoB,EAAE,CAAA;IAEjC,KAAK,MAAM,GAAG,IAAI,SAAS,CAAC,KAAK,EAAE,QAAQ,CAAC,EAAE,CAAC;QAC7C,0EAA0E;QAC1E,4EAA4E;QAC5E,qBAAqB;QACrB,MAAM,IAAI,GAAG,GAAG,CAAC,cAAc,CAAC,CAAA;QAChC,IAAI,IAAI,KAAK,EAAE,IAAI,CAAC,IAAI,CAAC,UAAU,CAAC,MAAM,CAAC;YAAE,SAAQ;QAErD,MAAM,KAAK,GAAG,GAAG,CAAC,kBAAkB,CAAC,IAAI,GAAG,CAAC,mBAAmB,CAAC,IAAI,GAAG,CAAA;QACxE,MAAM,KAAK,GAAG,MAAM,CAAC,GAAG,CAAC,iBAAiB,CAAC,IAAI,GAAG,CAAC,kBAAkB,CAAC,CAAC,CAAA;QACvE,MAAM,GAAG,GAAG,MAAM,CAAC,GAAG,CAAC,iBAAiB,CAAC,IAAI,GAAG,CAAC,kBAAkB,CAAC,CAAC,CAAA;QACrE,IAAI,CAAC,MAAM,CAAC,QAAQ,CAAC,KAAK,CAAC,IAAI,CAAC,MAAM,CAAC,QAAQ,CAAC,GAAG,CAAC;YAAE,SAAQ;QAC9D,KAAK,CAAC,IAAI,CAAC,EAAE,KAAK,EAAE,KAAK,EAAE,GAAG,EAAE,IAAI,EAAE,CAAC,CAAA;IACzC,CAAC;IAED,OAAO,KAAK,CAAA;AACd,CAAC;AAED,wDAAwD;AACxD,SAAS,UAAU,CAAC,KAAe;IACjC,KAAK,IAAI,CAAC,GAAG,CAAC,EAAE,CAAC,GAAG,KAAK,CAAC,MAAM,EAAE,CAAC,EAAE,EAAE,CAAC;QACtC,MAAM,IAAI,GAAG,KAAK,CAAC,CAAC,CAAE,CAAA;QACtB,IAAI,CAAC,IAAI,CAAC,UAAU,CAAC,eAAe,CAAC;YAAE,SAAQ;QAC/C,MAAM,MAAM,GAAG,IAAI,CAAC,KAAK,CAAC,eAAe,CAAC,MAAM,CAAC,CAAC,IAAI,EAAE,CAAA;QACxD,oEAAoE;QACpE,uEAAuE;QACvE,IAAI,MAAM,KAAK,EAAE;YAAE,OAAO,OAAO,CAAC,MAAM,CAAC,CAAA;QACzC,MAAM,IAAI,GAAG,KAAK,CAAC,CAAC,GAAG,CAAC,CAAC,EAAE,IAAI,EAAE,IAAI,EAAE,CAAA;QACvC,OAAO,OAAO,CAAC,IAAI,CAAC,UAAU,CAAC,GAAG,CAAC,CAAC,CAAC,CAAC,IAAI,CAAC,KAAK,CAAC,CAAC,CAAC,CAAC,CAAC,CAAC,IAAI,CAAC,CAAA;IAC7D,CAAC;IACD,OAAO,EAAE,CAAA;AACX,CAAC;AAED;;;;;;;;;;GAUG;AACH,QAAQ,CAAC,CAAC,SAAS,CACjB,KAAe,EACf,QAAgB;IAEhB,MAAM,MAAM,GAAG,GAAG,QAAQ,GAAG,CAAA;IAE7B,KAAK,IAAI,CAAC,GAAG,CAAC,EAAE,CAAC,GAAG,KAAK,CAAC,MAAM,EAAE,CAAC,EAAE,EAAE,CAAC;QACtC,MAAM,IAAI,GAAG,KAAK,CAAC,CAAC,CAAE,CAAC,IAAI,EAAE,CAAA;QAE7B,wEAAwE;QACxE,IAAI,IAAI,CAAC,UAAU,CAAC,MAAM,CAAC,EAAE,CAAC;YAC5B,MAAM,MAAM,GAAG,IAAI,GAAG,EAAkB,CAAA;YACxC,OAAO,CAAC,GAAG,KAAK,CAAC,MAAM,EAAE,CAAC;gBACxB,MAAM,IAAI,GAAG,KAAK,CAAC,CAAC,CAAE,CAAC,IAAI,EAAE,CAAA;gBAC7B,IAAI,CAAC,IAAI,CAAC,UAAU,CAAC,MAAM,CAAC;oBAAE,MAAK;gBACnC,MAAM,GAAG,GAAG,IAAI,CAAC,MAAM,CAAC,IAAI,CAAC,CAAA;gBAC7B,IAAI,GAAG,KAAK,CAAC,CAAC,EAAE,CAAC;oBACf,gDAAgD;oBAChD,MAAM,CAAC,GAAG,CAAC,IAAI,CAAC,KAAK,CAAC,MAAM,CAAC,MAAM,CAAC,EAAE,OAAO,CAAC,KAAK,CAAC,EAAE,CAAC,CAAC,EAAE,IAAI,EAAE,IAAI,EAAE,CAAC,CAAC,CAAA;gBAC1E,CAAC;qBAAM,CAAC;oBACN,MAAM,CAAC,GAAG,CACR,IAAI,CAAC,KAAK,CAAC,MAAM,CAAC,MAAM,EAAE,GAAG,CAAC,EAC9B,OAAO,CAAC,IAAI,CAAC,KAAK,CAAC,GAAG,CAAC,CAAC,IAAI,EAAE,CAAC,CAChC,CAAA;gBACH,CAAC;gBACD,CAAC,EAAE,CAAA;YACL,CAAC;YACD,MAAM,CAAC,KAAK,EAAE,EAAE,CAAC,MAAM,CAAC,GAAG,CAAC,KAAK,CAAC,IAAI,EAAE,CAAA;YACxC,OAAM;QACR,CAAC;QAED,IAAI,IAAI,KAAK,OAAO;YAAE,SAAQ;QAE9B,0EAA0E;QAC1E,kBAAkB;QAClB,MAAM,OAAO,GAAG,IAAI,GAAG,EAAkB,CAAA;QACzC,IAAI,MAAM,GAAG,CAAC,GAAG,CAAC,CAAA;QAClB,OAAO,MAAM,GAAG,KAAK,CAAC,MAAM,EAAE,CAAC;YAC7B,MAAM,MAAM,GAAG,KAAK,CAAC,MAAM,CAAE,CAAC,IAAI,EAAE,CAAA;YACpC,IAAI,CAAC,MAAM,CAAC,UAAU,CAAC,GAAG,CAAC;gBAAE,MAAK;YAClC,IAAI,MAAM,CAAC,UAAU,CAAC,MAAM,CAAC,EAAE,CAAC;gBAC9B,OAAO,CAAC,GAAG,CAAC,MAAM,CAAC,KAAK,CAAC,MAAM,CAAC,MAAM,CAAC,EAAE,OAAO,CAAC,IAAI,CAAC,CAAA;YACxD,CAAC;YACD,MAAM,EAAE,CAAA;QACV,CAAC;QACD,0EAA0E;QAC1E,oEAAoE;QACpE,IAAI,OAAO,CAAC,IAAI,KAAK,CAAC;YAAE,SAAQ;QAEhC,OAAO,MAAM,GAAG,KAAK,CAAC,MAAM,EAAE,MAAM,EAAE,EAAE,CAAC;YACvC,MAAM,GAAG,GAAG,KAAK,CAAC,MAAM,CAAE,CAAA;YAC1B,MAAM,OAAO,GAAG,GAAG,CAAC,IAAI,EAAE,CAAA;YAC1B,iEAAiE;YACjE,IAAI,OAAO,KAAK,EAAE,IAAI,OAAO,KAAK,GAAG,IAAI,OAAO,CAAC,UAAU,CAAC,GAAG,CAAC;gBAAE,MAAK;YACvE,IAAI,OAAO,KAAK,OAAO,IAAI,OAAO,CAAC,UAAU,CAAC,OAAO,CAAC;gBAAE,MAAK;YAE7D,MAAM,MAAM,GAAG,WAAW,CAAC,OAAO,CAAC,CAAA;YACnC,MAAM,CAAC,KAAK,EAAE,EAAE;gBACd,MAAM,KAAK,GAAG,OAAO,CAAC,GAAG,CAAC,KAAK,CAAC,CAAA;gBAChC,IAAI,KAAK,KAAK,SAAS;oBAAE,OAAO,EAAE,CAAA;gBAClC,MAAM,KAAK,GAAG,MAAM,CAAC,KAAK,CAAC,IAAI,EAAE,CAAA;gBACjC,OAAO,KAAK,KAAK,GAAG,IAAI,KAAK,KAAK,GAAG,CAAC,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC,KAAK,CAAA;YACpD,CAAC,CAAA;QACH,CAAC;QACD,OAAM;IACR,CAAC;AACH,CAAC;AAED;;;;;;GAMG;AACH,SAAS,WAAW,CAAC,GAAW;IAC9B,MAAM,MAAM,GAAa,EAAE,CAAA;IAC3B,IAAI,CAAC,GAAG,CAAC,CAAA;IAET,OAAO,CAAC,GAAG,GAAG,CAAC,MAAM,EAAE,CAAC;QACtB,MAAM,IAAI,GAAG,GAAG,CAAC,CAAC,CAAE,CAAA;QACpB,IAAI,IAAI,KAAK,GAAG,IAAI,IAAI,KAAK,IAAI,EAAE,CAAC;YAClC,CAAC,EAAE,CAAA;YACH,SAAQ;QACV,CAAC;QACD,IAAI,IAAI,KAAK,GAAG,IAAI,IAAI,KAAK,GAAG,EAAE,CAAC;YACjC,MAAM,GAAG,GAAG,GAAG,CAAC,OAAO,CAAC,IAAI,EAAE,CAAC,GAAG,CAAC,CAAC,CAAA;YACpC,IAAI,GAAG,KAAK,CAAC,CAAC,EAAE,CAAC;gBACf,MAAM,CAAC,IAAI,CAAC,GAAG,CAAC,KAAK,CAAC,CAAC,GAAG,CAAC,CAAC,CAAC,CAAA;gBAC7B,MAAK;YACP,CAAC;YACD,MAAM,CAAC,IAAI,CAAC,GAAG,CAAC,KAAK,CAAC,CAAC,GAAG,CAAC,EAAE,GAAG,CAAC,CAAC,CAAA;YAClC,CAAC,GAAG,GAAG,GAAG,CAAC,CAAA;YACX,SAAQ;QACV,CAAC;QACD,IAAI,GAAG,GAAG,CAAC,CAAA;QACX,OAAO,GAAG,GAAG,GAAG,CAAC,MAAM,IAAI,GAAG,CAAC,GAAG,CAAC,KAAK,GAAG,IAAI,GAAG,CAAC,GAAG,CAAC,KAAK,IAAI;YAAE,GAAG,EAAE,CAAA;QACvE,MAAM,CAAC,IAAI,CAAC,GAAG,CAAC,KAAK,CAAC,CAAC,EAAE,GAAG,CAAC,CAAC,CAAA;QAC9B,CAAC,GAAG,GAAG,CAAA;IACT,CAAC;IAED,OAAO,MAAM,CAAA;AACf,CAAC;AAED,2EAA2E;AAC3E,SAAS,OAAO,CAAC,KAAa;IAC5B,MAAM,IAAI,GAAG,KAAK,CAAC,IAAI,EAAE,CAAA;IACzB,IAAI,IAAI,KAAK,GAAG,IAAI,IAAI,KAAK,GAAG;QAAE,OAAO,EAAE,CAAA;IAC3C,MAAM,KAAK,GAAG,IAAI,CAAC,CAAC,CAAC,CAAA;IACrB,IAAI,CAAC,KAAK,KAAK,GAAG,IAAI,KAAK,KAAK,GAAG,CAAC,IAAI,IAAI,CAAC,QAAQ,CAAC,KAAK,CAAC,IAAI,IAAI,CAAC,MAAM,GAAG,CAAC,EAAE,CAAC;QAChF,OAAO,IAAI,CAAC,KAAK,CAAC,CAAC,EAAE,CAAC,CAAC,CAAC,CAAA;IAC1B,CAAC;IACD,OAAO,IAAI,CAAA;AACb,CAAC","sourcesContent":["/**\n * Reads a coordinate file into a {@link ProteinStructure}.\n *\n * Two formats, because the Protein Data Bank serves two and which one an entry\n * has is not the author's choice: the legacy **PDB** format numbers atoms in\n * five columns and names chains in one, so an entry that outgrew either — a\n * ribosome, a capsid — exists only as **mmCIF**. Refusing the second would mean\n * refusing exactly the structures most worth looking at.\n *\n * Both are read the same way: pull out an atom list and whatever secondary\n * structure the file annotates, then hand both to `deriveStructure`, which does\n * the geometry. {@link parseStructure} sniffs which is which, so a caller\n * holding a downloaded file never has to.\n *\n * **The first model only.** An NMR ensemble is twenty superposed conformations\n * of the same molecule; drawing all of them at once produces a blur, and\n * choosing between them is a question this node does not ask. The first is the\n * conventional representative.\n */\n\nimport { elementOf, residueKind } from \"./chemistry\"\nimport {\n deriveStructure,\n EMPTY_STRUCTURE,\n type ProteinAtom,\n type ProteinStructure,\n type SecondarySpan,\n} from \"./structure\"\n\n/**\n * Reads a coordinate file, detecting its format from the contents.\n *\n * `id` labels the result — an accession, or the asset's name. Anything that\n * can't be read at all comes back as {@link EMPTY_STRUCTURE} rather than\n * throwing: this runs against a file someone just uploaded or an accession\n * someone is halfway through typing, and both of those are ordinary states\n * rather than errors. An empty structure draws nothing, which is the honest\n * picture of a file with no atoms in it.\n */\nexport function parseStructure(text: string, id = \"\"): ProteinStructure {\n if (typeof text !== \"string\" || text.trim() === \"\") {\n return { ...EMPTY_STRUCTURE, id }\n }\n return isMmcif(text) ? parseMmcif(text, id) : parsePdb(text, id)\n}\n\n/**\n * Whether a file is mmCIF.\n *\n * The `data_` block header is mmCIF's first line and appears in no PDB file, so\n * it decides on its own — but only over the first stretch, since `data_` is also\n * an ordinary substring that could turn up in a PDB `REMARK`. The category\n * prefix is the belt-and-braces check for a fragment handed over without its\n * header.\n */\nfunction isMmcif(text: string): boolean {\n const head = text.slice(0, 4096)\n return /^\\s*data_/.test(head) || head.includes(\"_atom_site.\")\n}\n\n// --- PDB -------------------------------------------------------------------\n\n/**\n * Reads the legacy PDB format, which is **fixed-column**: every field is at a\n * known offset and the whitespace between them is padding rather than a\n * separator. Splitting on spaces is the classic way to get this wrong — a\n * residue number that runs into its insertion code, or a `-` sign that eats the\n * gap between two coordinates, and the record silently comes apart.\n *\n * Columns, 1-based as the specification numbers them:\n *\n * 7–11 serial 13–16 name 18–20 resName 22 chainID\n * 23–26 resSeq 31–38 x 39–46 y 47–54 z\n * 77–78 element\n */\nfunction parsePdb(text: string, id: string): ProteinStructure {\n const atoms: ProteinAtom[] = []\n const spans: SecondarySpan[] = []\n let title = \"\"\n\n for (const line of text.split(\"\\n\")) {\n const record = line.slice(0, 6).trim()\n\n // Everything after the first `ENDMDL` is another conformation of what we\n // already have. Stopping at it rather than filtering by model number also\n // ends the read early on a large ensemble.\n if (record === \"ENDMDL\") break\n\n if (record === \"ATOM\" || record === \"HETATM\") {\n const atom = parsePdbAtom(line, record === \"HETATM\")\n if (atom) atoms.push(atom)\n continue\n }\n\n if (record === \"TITLE\") {\n // A long title is continued across several records, with the continuation\n // number in columns 9–10 and the text always from column 11.\n title = `${title} ${line.slice(10).trim()}`.trim()\n continue\n }\n\n // The two annotation records. Their first-residue fields sit at *different*\n // offsets from each other — a genuine wart of the format rather than a\n // mistake here — while their last-residue fields agree.\n if (record === \"HELIX\") {\n const span = pdbSpan(line, 19, 21, \"helix\")\n if (span) spans.push(span)\n continue\n }\n if (record === \"SHEET\") {\n const span = pdbSpan(line, 21, 22, \"sheet\")\n if (span) spans.push(span)\n }\n }\n\n return deriveStructure(id, title, atoms, spans)\n}\n\n/** One `ATOM`/`HETATM` record, or `null` when its coordinates don't read. */\nfunction parsePdbAtom(line: string, hetero: boolean): ProteinAtom | null {\n // Truncated before the coordinates. Worth checking outright: a short slice of\n // a fixed-column record reads as the empty string, and `Number(\"\")` is 0\n // rather than `NaN` — so without this a mangled line becomes an atom sitting\n // at the origin, which is far harder to notice than a missing one.\n if (line.length < 54) return null\n\n // An alternate location: the same atom modelled twice because the side chain\n // is disordered. Keeping both would double the sticks through that residue,\n // so take the first (blank or `A`), which is the convention.\n const altLoc = line.slice(16, 17).trim()\n if (altLoc !== \"\" && altLoc !== \"A\") return null\n\n const x = Number(line.slice(30, 38))\n const y = Number(line.slice(38, 46))\n const z = Number(line.slice(46, 54))\n if (!Number.isFinite(x) || !Number.isFinite(y) || !Number.isFinite(z)) {\n return null\n }\n\n const name = line.slice(12, 16).trim()\n const residue = line.slice(17, 20).trim()\n const kind = residueKind(residue)\n\n return {\n serial: Number(line.slice(6, 11)) || 0,\n name,\n element: elementOf(line.slice(76, 78), name, kind),\n residue,\n residueSeq: Number(line.slice(22, 26)) || 0,\n // A file with a single unnamed chain leaves the column blank; calling that\n // `A` keeps \"group by chain\" from producing one group named nothing.\n chain: line.slice(21, 22).trim() || \"A\",\n x,\n y,\n z,\n hetero,\n kind,\n }\n}\n\n/** Where the last residue's number sits on both `HELIX` and `SHEET` — columns 34–37. */\nconst PDB_SPAN_END_AT = 33\n\n/**\n * One `HELIX`/`SHEET` record as a residue span.\n *\n * The first-residue offsets differ between the two records, so they are passed\n * in rather than hard-coded: a helix names its chain in column 20 and its first\n * residue in 22–25, a sheet names its chain in 22 and its first residue in\n * 23–26. Their *last* residue is in the same place in both, which is the one\n * thing they agree on.\n */\nfunction pdbSpan(\n line: string,\n chainAt: number,\n startAt: number,\n kind: SecondarySpan[\"kind\"]\n): SecondarySpan | null {\n const chain = line.slice(chainAt, chainAt + 1).trim() || \"A\"\n // Read as text first: a blank fixed-column field slices to `\"\"`, and\n // `Number(\"\")` is 0, so a `Number.isFinite` check alone would accept a record\n // that states nothing as a span over residue 0.\n const startText = line.slice(startAt, startAt + 4).trim()\n const endText = line.slice(PDB_SPAN_END_AT, PDB_SPAN_END_AT + 4).trim()\n if (startText === \"\" || endText === \"\") return null\n\n const start = Number(startText)\n const end = Number(endText)\n if (!Number.isFinite(start) || !Number.isFinite(end)) return null\n return { chain, start, end, kind }\n}\n\n// --- mmCIF -----------------------------------------------------------------\n\n/**\n * Reads mmCIF, which is a tagged format rather than a positional one: values are\n * whitespace-separated and their meaning comes from the column headers declared\n * above them, so the reader has to learn the layout before it can read a row.\n *\n * Only three categories are consulted — the atoms, and the two that annotate\n * secondary structure. mmCIF carries dozens more, and every one this doesn't\n * read is one this can't be broken by.\n */\nfunction parseMmcif(text: string, id: string): ProteinStructure {\n const lines = text.split(\"\\n\")\n const atoms = mmcifAtoms(lines)\n const spans = [\n ...mmcifSpans(lines, \"_struct_conf\", \"helix\"),\n ...mmcifSpans(lines, \"_struct_sheet_range\", \"sheet\"),\n ]\n return deriveStructure(id, mmcifTitle(lines), atoms, spans)\n}\n\n/** The atoms of the first model in an `_atom_site` loop. */\nfunction mmcifAtoms(lines: string[]): ProteinAtom[] {\n const atoms: ProteinAtom[] = []\n let model: string | null = null\n\n for (const row of mmcifRows(lines, \"_atom_site\")) {\n // `auth_*` is the numbering the literature uses and the one a PDB file would\n // have carried; `label_*` is the internal, re-derived scheme. Preferring\n // auth means a residue number quoted in a paper matches what is drawn, and\n // falling back keeps a file that omits it readable.\n const chain = row(\"auth_asym_id\") || row(\"label_asym_id\") || \"A\"\n const residue = row(\"auth_comp_id\") || row(\"label_comp_id\")\n const name = row(\"auth_atom_id\") || row(\"label_atom_id\")\n\n const thisModel = row(\"pdbx_PDB_model_num\")\n if (model === null) model = thisModel\n else if (thisModel !== model) break\n\n const altLoc = row(\"label_alt_id\")\n if (altLoc !== \"\" && altLoc !== \".\" && altLoc !== \"?\" && altLoc !== \"A\") {\n continue\n }\n\n const x = Number(row(\"Cartn_x\"))\n const y = Number(row(\"Cartn_y\"))\n const z = Number(row(\"Cartn_z\"))\n if (!Number.isFinite(x) || !Number.isFinite(y) || !Number.isFinite(z)) {\n continue\n }\n\n const kind = residueKind(residue)\n atoms.push({\n serial: Number(row(\"id\")) || 0,\n name,\n element: elementOf(row(\"type_symbol\"), name, kind),\n residue,\n residueSeq: Number(row(\"auth_seq_id\") || row(\"label_seq_id\")) || 0,\n chain,\n x,\n y,\n z,\n hetero: row(\"group_PDB\") === \"HETATM\",\n kind,\n })\n }\n\n return atoms\n}\n\n/** The residue spans of one annotation category. */\nfunction mmcifSpans(\n lines: string[],\n category: string,\n kind: SecondarySpan[\"kind\"]\n): SecondarySpan[] {\n const spans: SecondarySpan[] = []\n\n for (const row of mmcifRows(lines, category)) {\n // `_struct_conf` covers turns and bends as well as helices, all under one\n // category and told apart by this tag. A sheet range has no such tag, so an\n // absent one passes.\n const type = row(\"conf_type_id\")\n if (type !== \"\" && !type.startsWith(\"HELX\")) continue\n\n const chain = row(\"beg_auth_asym_id\") || row(\"beg_label_asym_id\") || \"A\"\n const start = Number(row(\"beg_auth_seq_id\") || row(\"beg_label_seq_id\"))\n const end = Number(row(\"end_auth_seq_id\") || row(\"end_label_seq_id\"))\n if (!Number.isFinite(start) || !Number.isFinite(end)) continue\n spans.push({ chain, start, end, kind })\n }\n\n return spans\n}\n\n/** The entry's title, from the `_struct.title` item. */\nfunction mmcifTitle(lines: string[]): string {\n for (let i = 0; i < lines.length; i++) {\n const line = lines[i]!\n if (!line.startsWith(\"_struct.title\")) continue\n const inline = line.slice(\"_struct.title\".length).trim()\n // A value too long for the line is carried on the next one, or in a\n // semicolon-delimited block below it; the inline form covers the rest.\n if (inline !== \"\") return unquote(inline)\n const next = lines[i + 1]?.trim() ?? \"\"\n return unquote(next.startsWith(\";\") ? next.slice(1) : next)\n }\n return \"\"\n}\n\n/**\n * Walks the rows of one mmCIF category, yielding a field reader for each.\n *\n * The reader is a closure over the current row rather than an object, so\n * nothing allocates a record per atom — this runs over hundreds of thousands of\n * rows on a large entry, and the caller only ever wants a handful of the fields.\n *\n * Handles both forms a category takes: a `loop_` with a header block and many\n * rows, and the flat `_category.field value` form mmCIF uses when there is\n * exactly one row.\n */\nfunction* mmcifRows(\n lines: string[],\n category: string\n): Generator<(field: string) => string> {\n const prefix = `${category}.`\n\n for (let i = 0; i < lines.length; i++) {\n const line = lines[i]!.trim()\n\n // The flat, single-row form: consecutive `_category.field value` lines.\n if (line.startsWith(prefix)) {\n const single = new Map<string, string>()\n while (i < lines.length) {\n const item = lines[i]!.trim()\n if (!item.startsWith(prefix)) break\n const gap = item.search(/\\s/)\n if (gap === -1) {\n // A field whose value is on the following line.\n single.set(item.slice(prefix.length), unquote(lines[++i]?.trim() ?? \"\"))\n } else {\n single.set(\n item.slice(prefix.length, gap),\n unquote(item.slice(gap).trim())\n )\n }\n i++\n }\n yield (field) => single.get(field) ?? \"\"\n return\n }\n\n if (line !== \"loop_\") continue\n\n // The header block: every `_category.field` line, in the order the values\n // will arrive in.\n const columns = new Map<string, number>()\n let cursor = i + 1\n while (cursor < lines.length) {\n const header = lines[cursor]!.trim()\n if (!header.startsWith(\"_\")) break\n if (header.startsWith(prefix)) {\n columns.set(header.slice(prefix.length), columns.size)\n }\n cursor++\n }\n // A `loop_` for some other category. Its rows carry no leading underscore\n // and aren't `loop_`, so the outer scan walks past them harmlessly.\n if (columns.size === 0) continue\n\n for (; cursor < lines.length; cursor++) {\n const row = lines[cursor]!\n const trimmed = row.trim()\n // A loop ends at the next block, the next loop, or a blank line.\n if (trimmed === \"\" || trimmed === \"#\" || trimmed.startsWith(\"_\")) break\n if (trimmed === \"loop_\" || trimmed.startsWith(\"data_\")) break\n\n const values = splitCifRow(trimmed)\n yield (field) => {\n const index = columns.get(field)\n if (index === undefined) return \"\"\n const value = values[index] ?? \"\"\n return value === \".\" || value === \"?\" ? \"\" : value\n }\n }\n return\n }\n}\n\n/**\n * Splits one mmCIF row into values, respecting quotes.\n *\n * Quoting is not decoration here: a chain identifier can be `'A'`, and a residue\n * name can contain a space that a plain `split` would turn into two columns and\n * shift every field after it by one.\n */\nfunction splitCifRow(row: string): string[] {\n const values: string[] = []\n let i = 0\n\n while (i < row.length) {\n const char = row[i]!\n if (char === \" \" || char === \"\\t\") {\n i++\n continue\n }\n if (char === \"'\" || char === '\"') {\n const end = row.indexOf(char, i + 1)\n if (end === -1) {\n values.push(row.slice(i + 1))\n break\n }\n values.push(row.slice(i + 1, end))\n i = end + 1\n continue\n }\n let end = i\n while (end < row.length && row[end] !== \" \" && row[end] !== \"\\t\") end++\n values.push(row.slice(i, end))\n i = end\n }\n\n return values\n}\n\n/** Strips the quotes mmCIF wraps a value in, and the null placeholders. */\nfunction unquote(value: string): string {\n const text = value.trim()\n if (text === \".\" || text === \"?\") return \"\"\n const first = text[0]\n if ((first === \"'\" || first === '\"') && text.endsWith(first) && text.length > 1) {\n return text.slice(1, -1)\n }\n return text\n}\n"]}