@motionscript/molecule 0.0.0-stage → 0.1.0-alpha.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +5 -0
- package/LICENSE +201 -0
- package/dist/browser/index.js +4 -0
- package/dist/browser/index.js.map +7 -0
- package/dist/browser/manifest.json +11 -0
- package/dist/index.d.ts +3 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +3 -0
- package/dist/index.js.map +1 -0
- package/dist/nodes.d.ts +18 -0
- package/dist/nodes.d.ts.map +1 -0
- package/dist/nodes.js +18 -0
- package/dist/nodes.js.map +1 -0
- package/dist/protein/chemistry.d.ts +52 -0
- package/dist/protein/chemistry.d.ts.map +1 -0
- package/dist/protein/chemistry.js +208 -0
- package/dist/protein/chemistry.js.map +1 -0
- package/dist/protein/index.d.ts +35 -0
- package/dist/protein/index.d.ts.map +1 -0
- package/dist/protein/index.js +35 -0
- package/dist/protein/index.js.map +1 -0
- package/dist/protein/parse.d.ts +32 -0
- package/dist/protein/parse.d.ts.map +1 -0
- package/dist/protein/parse.js +387 -0
- package/dist/protein/parse.js.map +1 -0
- package/dist/protein/protein.d.ts +265 -0
- package/dist/protein/protein.d.ts.map +1 -0
- package/dist/protein/protein.js +645 -0
- package/dist/protein/protein.js.map +1 -0
- package/dist/protein/ribbon.d.ts +83 -0
- package/dist/protein/ribbon.d.ts.map +1 -0
- package/dist/protein/ribbon.js +468 -0
- package/dist/protein/ribbon.js.map +1 -0
- package/dist/protein/shared.d.ts +221 -0
- package/dist/protein/shared.d.ts.map +1 -0
- package/dist/protein/shared.js +478 -0
- package/dist/protein/shared.js.map +1 -0
- package/dist/protein/structure.d.ts +184 -0
- package/dist/protein/structure.d.ts.map +1 -0
- package/dist/protein/structure.js +324 -0
- package/dist/protein/structure.js.map +1 -0
- package/package.json +64 -3
- package/registry.json +22 -0
- package/src/index.ts +2 -0
- package/src/nodes.ts +18 -0
- package/src/protein/chemistry.ts +223 -0
- package/src/protein/index.ts +34 -0
- package/src/protein/parse.ts +427 -0
- package/src/protein/protein.ts +897 -0
- package/src/protein/ribbon.ts +658 -0
- package/src/protein/shared.ts +622 -0
- package/src/protein/structure.ts +491 -0
- package/README.md +0 -4
|
@@ -0,0 +1,427 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Reads a coordinate file into a {@link ProteinStructure}.
|
|
3
|
+
*
|
|
4
|
+
* Two formats, because the Protein Data Bank serves two and which one an entry
|
|
5
|
+
* has is not the author's choice: the legacy **PDB** format numbers atoms in
|
|
6
|
+
* five columns and names chains in one, so an entry that outgrew either — a
|
|
7
|
+
* ribosome, a capsid — exists only as **mmCIF**. Refusing the second would mean
|
|
8
|
+
* refusing exactly the structures most worth looking at.
|
|
9
|
+
*
|
|
10
|
+
* Both are read the same way: pull out an atom list and whatever secondary
|
|
11
|
+
* structure the file annotates, then hand both to `deriveStructure`, which does
|
|
12
|
+
* the geometry. {@link parseStructure} sniffs which is which, so a caller
|
|
13
|
+
* holding a downloaded file never has to.
|
|
14
|
+
*
|
|
15
|
+
* **The first model only.** An NMR ensemble is twenty superposed conformations
|
|
16
|
+
* of the same molecule; drawing all of them at once produces a blur, and
|
|
17
|
+
* choosing between them is a question this node does not ask. The first is the
|
|
18
|
+
* conventional representative.
|
|
19
|
+
*/
|
|
20
|
+
|
|
21
|
+
import { elementOf, residueKind } from "./chemistry"
|
|
22
|
+
import {
|
|
23
|
+
deriveStructure,
|
|
24
|
+
EMPTY_STRUCTURE,
|
|
25
|
+
type ProteinAtom,
|
|
26
|
+
type ProteinStructure,
|
|
27
|
+
type SecondarySpan,
|
|
28
|
+
} from "./structure"
|
|
29
|
+
|
|
30
|
+
/**
|
|
31
|
+
* Reads a coordinate file, detecting its format from the contents.
|
|
32
|
+
*
|
|
33
|
+
* `id` labels the result — an accession, or the asset's name. Anything that
|
|
34
|
+
* can't be read at all comes back as {@link EMPTY_STRUCTURE} rather than
|
|
35
|
+
* throwing: this runs against a file someone just uploaded or an accession
|
|
36
|
+
* someone is halfway through typing, and both of those are ordinary states
|
|
37
|
+
* rather than errors. An empty structure draws nothing, which is the honest
|
|
38
|
+
* picture of a file with no atoms in it.
|
|
39
|
+
*/
|
|
40
|
+
export function parseStructure(text: string, id = ""): ProteinStructure {
|
|
41
|
+
if (typeof text !== "string" || text.trim() === "") {
|
|
42
|
+
return { ...EMPTY_STRUCTURE, id }
|
|
43
|
+
}
|
|
44
|
+
return isMmcif(text) ? parseMmcif(text, id) : parsePdb(text, id)
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
/**
|
|
48
|
+
* Whether a file is mmCIF.
|
|
49
|
+
*
|
|
50
|
+
* The `data_` block header is mmCIF's first line and appears in no PDB file, so
|
|
51
|
+
* it decides on its own — but only over the first stretch, since `data_` is also
|
|
52
|
+
* an ordinary substring that could turn up in a PDB `REMARK`. The category
|
|
53
|
+
* prefix is the belt-and-braces check for a fragment handed over without its
|
|
54
|
+
* header.
|
|
55
|
+
*/
|
|
56
|
+
function isMmcif(text: string): boolean {
|
|
57
|
+
const head = text.slice(0, 4096)
|
|
58
|
+
return /^\s*data_/.test(head) || head.includes("_atom_site.")
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
// --- PDB -------------------------------------------------------------------
|
|
62
|
+
|
|
63
|
+
/**
|
|
64
|
+
* Reads the legacy PDB format, which is **fixed-column**: every field is at a
|
|
65
|
+
* known offset and the whitespace between them is padding rather than a
|
|
66
|
+
* separator. Splitting on spaces is the classic way to get this wrong — a
|
|
67
|
+
* residue number that runs into its insertion code, or a `-` sign that eats the
|
|
68
|
+
* gap between two coordinates, and the record silently comes apart.
|
|
69
|
+
*
|
|
70
|
+
* Columns, 1-based as the specification numbers them:
|
|
71
|
+
*
|
|
72
|
+
* 7–11 serial 13–16 name 18–20 resName 22 chainID
|
|
73
|
+
* 23–26 resSeq 31–38 x 39–46 y 47–54 z
|
|
74
|
+
* 77–78 element
|
|
75
|
+
*/
|
|
76
|
+
function parsePdb(text: string, id: string): ProteinStructure {
|
|
77
|
+
const atoms: ProteinAtom[] = []
|
|
78
|
+
const spans: SecondarySpan[] = []
|
|
79
|
+
let title = ""
|
|
80
|
+
|
|
81
|
+
for (const line of text.split("\n")) {
|
|
82
|
+
const record = line.slice(0, 6).trim()
|
|
83
|
+
|
|
84
|
+
// Everything after the first `ENDMDL` is another conformation of what we
|
|
85
|
+
// already have. Stopping at it rather than filtering by model number also
|
|
86
|
+
// ends the read early on a large ensemble.
|
|
87
|
+
if (record === "ENDMDL") break
|
|
88
|
+
|
|
89
|
+
if (record === "ATOM" || record === "HETATM") {
|
|
90
|
+
const atom = parsePdbAtom(line, record === "HETATM")
|
|
91
|
+
if (atom) atoms.push(atom)
|
|
92
|
+
continue
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
if (record === "TITLE") {
|
|
96
|
+
// A long title is continued across several records, with the continuation
|
|
97
|
+
// number in columns 9–10 and the text always from column 11.
|
|
98
|
+
title = `${title} ${line.slice(10).trim()}`.trim()
|
|
99
|
+
continue
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
// The two annotation records. Their first-residue fields sit at *different*
|
|
103
|
+
// offsets from each other — a genuine wart of the format rather than a
|
|
104
|
+
// mistake here — while their last-residue fields agree.
|
|
105
|
+
if (record === "HELIX") {
|
|
106
|
+
const span = pdbSpan(line, 19, 21, "helix")
|
|
107
|
+
if (span) spans.push(span)
|
|
108
|
+
continue
|
|
109
|
+
}
|
|
110
|
+
if (record === "SHEET") {
|
|
111
|
+
const span = pdbSpan(line, 21, 22, "sheet")
|
|
112
|
+
if (span) spans.push(span)
|
|
113
|
+
}
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
return deriveStructure(id, title, atoms, spans)
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
/** One `ATOM`/`HETATM` record, or `null` when its coordinates don't read. */
|
|
120
|
+
function parsePdbAtom(line: string, hetero: boolean): ProteinAtom | null {
|
|
121
|
+
// Truncated before the coordinates. Worth checking outright: a short slice of
|
|
122
|
+
// a fixed-column record reads as the empty string, and `Number("")` is 0
|
|
123
|
+
// rather than `NaN` — so without this a mangled line becomes an atom sitting
|
|
124
|
+
// at the origin, which is far harder to notice than a missing one.
|
|
125
|
+
if (line.length < 54) return null
|
|
126
|
+
|
|
127
|
+
// An alternate location: the same atom modelled twice because the side chain
|
|
128
|
+
// is disordered. Keeping both would double the sticks through that residue,
|
|
129
|
+
// so take the first (blank or `A`), which is the convention.
|
|
130
|
+
const altLoc = line.slice(16, 17).trim()
|
|
131
|
+
if (altLoc !== "" && altLoc !== "A") return null
|
|
132
|
+
|
|
133
|
+
const x = Number(line.slice(30, 38))
|
|
134
|
+
const y = Number(line.slice(38, 46))
|
|
135
|
+
const z = Number(line.slice(46, 54))
|
|
136
|
+
if (!Number.isFinite(x) || !Number.isFinite(y) || !Number.isFinite(z)) {
|
|
137
|
+
return null
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
const name = line.slice(12, 16).trim()
|
|
141
|
+
const residue = line.slice(17, 20).trim()
|
|
142
|
+
const kind = residueKind(residue)
|
|
143
|
+
|
|
144
|
+
return {
|
|
145
|
+
serial: Number(line.slice(6, 11)) || 0,
|
|
146
|
+
name,
|
|
147
|
+
element: elementOf(line.slice(76, 78), name, kind),
|
|
148
|
+
residue,
|
|
149
|
+
residueSeq: Number(line.slice(22, 26)) || 0,
|
|
150
|
+
// A file with a single unnamed chain leaves the column blank; calling that
|
|
151
|
+
// `A` keeps "group by chain" from producing one group named nothing.
|
|
152
|
+
chain: line.slice(21, 22).trim() || "A",
|
|
153
|
+
x,
|
|
154
|
+
y,
|
|
155
|
+
z,
|
|
156
|
+
hetero,
|
|
157
|
+
kind,
|
|
158
|
+
}
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
/** Where the last residue's number sits on both `HELIX` and `SHEET` — columns 34–37. */
|
|
162
|
+
const PDB_SPAN_END_AT = 33
|
|
163
|
+
|
|
164
|
+
/**
|
|
165
|
+
* One `HELIX`/`SHEET` record as a residue span.
|
|
166
|
+
*
|
|
167
|
+
* The first-residue offsets differ between the two records, so they are passed
|
|
168
|
+
* in rather than hard-coded: a helix names its chain in column 20 and its first
|
|
169
|
+
* residue in 22–25, a sheet names its chain in 22 and its first residue in
|
|
170
|
+
* 23–26. Their *last* residue is in the same place in both, which is the one
|
|
171
|
+
* thing they agree on.
|
|
172
|
+
*/
|
|
173
|
+
function pdbSpan(
|
|
174
|
+
line: string,
|
|
175
|
+
chainAt: number,
|
|
176
|
+
startAt: number,
|
|
177
|
+
kind: SecondarySpan["kind"]
|
|
178
|
+
): SecondarySpan | null {
|
|
179
|
+
const chain = line.slice(chainAt, chainAt + 1).trim() || "A"
|
|
180
|
+
// Read as text first: a blank fixed-column field slices to `""`, and
|
|
181
|
+
// `Number("")` is 0, so a `Number.isFinite` check alone would accept a record
|
|
182
|
+
// that states nothing as a span over residue 0.
|
|
183
|
+
const startText = line.slice(startAt, startAt + 4).trim()
|
|
184
|
+
const endText = line.slice(PDB_SPAN_END_AT, PDB_SPAN_END_AT + 4).trim()
|
|
185
|
+
if (startText === "" || endText === "") return null
|
|
186
|
+
|
|
187
|
+
const start = Number(startText)
|
|
188
|
+
const end = Number(endText)
|
|
189
|
+
if (!Number.isFinite(start) || !Number.isFinite(end)) return null
|
|
190
|
+
return { chain, start, end, kind }
|
|
191
|
+
}
|
|
192
|
+
|
|
193
|
+
// --- mmCIF -----------------------------------------------------------------
|
|
194
|
+
|
|
195
|
+
/**
|
|
196
|
+
* Reads mmCIF, which is a tagged format rather than a positional one: values are
|
|
197
|
+
* whitespace-separated and their meaning comes from the column headers declared
|
|
198
|
+
* above them, so the reader has to learn the layout before it can read a row.
|
|
199
|
+
*
|
|
200
|
+
* Only three categories are consulted — the atoms, and the two that annotate
|
|
201
|
+
* secondary structure. mmCIF carries dozens more, and every one this doesn't
|
|
202
|
+
* read is one this can't be broken by.
|
|
203
|
+
*/
|
|
204
|
+
function parseMmcif(text: string, id: string): ProteinStructure {
|
|
205
|
+
const lines = text.split("\n")
|
|
206
|
+
const atoms = mmcifAtoms(lines)
|
|
207
|
+
const spans = [
|
|
208
|
+
...mmcifSpans(lines, "_struct_conf", "helix"),
|
|
209
|
+
...mmcifSpans(lines, "_struct_sheet_range", "sheet"),
|
|
210
|
+
]
|
|
211
|
+
return deriveStructure(id, mmcifTitle(lines), atoms, spans)
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
/** The atoms of the first model in an `_atom_site` loop. */
|
|
215
|
+
function mmcifAtoms(lines: string[]): ProteinAtom[] {
|
|
216
|
+
const atoms: ProteinAtom[] = []
|
|
217
|
+
let model: string | null = null
|
|
218
|
+
|
|
219
|
+
for (const row of mmcifRows(lines, "_atom_site")) {
|
|
220
|
+
// `auth_*` is the numbering the literature uses and the one a PDB file would
|
|
221
|
+
// have carried; `label_*` is the internal, re-derived scheme. Preferring
|
|
222
|
+
// auth means a residue number quoted in a paper matches what is drawn, and
|
|
223
|
+
// falling back keeps a file that omits it readable.
|
|
224
|
+
const chain = row("auth_asym_id") || row("label_asym_id") || "A"
|
|
225
|
+
const residue = row("auth_comp_id") || row("label_comp_id")
|
|
226
|
+
const name = row("auth_atom_id") || row("label_atom_id")
|
|
227
|
+
|
|
228
|
+
const thisModel = row("pdbx_PDB_model_num")
|
|
229
|
+
if (model === null) model = thisModel
|
|
230
|
+
else if (thisModel !== model) break
|
|
231
|
+
|
|
232
|
+
const altLoc = row("label_alt_id")
|
|
233
|
+
if (altLoc !== "" && altLoc !== "." && altLoc !== "?" && altLoc !== "A") {
|
|
234
|
+
continue
|
|
235
|
+
}
|
|
236
|
+
|
|
237
|
+
const x = Number(row("Cartn_x"))
|
|
238
|
+
const y = Number(row("Cartn_y"))
|
|
239
|
+
const z = Number(row("Cartn_z"))
|
|
240
|
+
if (!Number.isFinite(x) || !Number.isFinite(y) || !Number.isFinite(z)) {
|
|
241
|
+
continue
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
const kind = residueKind(residue)
|
|
245
|
+
atoms.push({
|
|
246
|
+
serial: Number(row("id")) || 0,
|
|
247
|
+
name,
|
|
248
|
+
element: elementOf(row("type_symbol"), name, kind),
|
|
249
|
+
residue,
|
|
250
|
+
residueSeq: Number(row("auth_seq_id") || row("label_seq_id")) || 0,
|
|
251
|
+
chain,
|
|
252
|
+
x,
|
|
253
|
+
y,
|
|
254
|
+
z,
|
|
255
|
+
hetero: row("group_PDB") === "HETATM",
|
|
256
|
+
kind,
|
|
257
|
+
})
|
|
258
|
+
}
|
|
259
|
+
|
|
260
|
+
return atoms
|
|
261
|
+
}
|
|
262
|
+
|
|
263
|
+
/** The residue spans of one annotation category. */
|
|
264
|
+
function mmcifSpans(
|
|
265
|
+
lines: string[],
|
|
266
|
+
category: string,
|
|
267
|
+
kind: SecondarySpan["kind"]
|
|
268
|
+
): SecondarySpan[] {
|
|
269
|
+
const spans: SecondarySpan[] = []
|
|
270
|
+
|
|
271
|
+
for (const row of mmcifRows(lines, category)) {
|
|
272
|
+
// `_struct_conf` covers turns and bends as well as helices, all under one
|
|
273
|
+
// category and told apart by this tag. A sheet range has no such tag, so an
|
|
274
|
+
// absent one passes.
|
|
275
|
+
const type = row("conf_type_id")
|
|
276
|
+
if (type !== "" && !type.startsWith("HELX")) continue
|
|
277
|
+
|
|
278
|
+
const chain = row("beg_auth_asym_id") || row("beg_label_asym_id") || "A"
|
|
279
|
+
const start = Number(row("beg_auth_seq_id") || row("beg_label_seq_id"))
|
|
280
|
+
const end = Number(row("end_auth_seq_id") || row("end_label_seq_id"))
|
|
281
|
+
if (!Number.isFinite(start) || !Number.isFinite(end)) continue
|
|
282
|
+
spans.push({ chain, start, end, kind })
|
|
283
|
+
}
|
|
284
|
+
|
|
285
|
+
return spans
|
|
286
|
+
}
|
|
287
|
+
|
|
288
|
+
/** The entry's title, from the `_struct.title` item. */
|
|
289
|
+
function mmcifTitle(lines: string[]): string {
|
|
290
|
+
for (let i = 0; i < lines.length; i++) {
|
|
291
|
+
const line = lines[i]!
|
|
292
|
+
if (!line.startsWith("_struct.title")) continue
|
|
293
|
+
const inline = line.slice("_struct.title".length).trim()
|
|
294
|
+
// A value too long for the line is carried on the next one, or in a
|
|
295
|
+
// semicolon-delimited block below it; the inline form covers the rest.
|
|
296
|
+
if (inline !== "") return unquote(inline)
|
|
297
|
+
const next = lines[i + 1]?.trim() ?? ""
|
|
298
|
+
return unquote(next.startsWith(";") ? next.slice(1) : next)
|
|
299
|
+
}
|
|
300
|
+
return ""
|
|
301
|
+
}
|
|
302
|
+
|
|
303
|
+
/**
|
|
304
|
+
* Walks the rows of one mmCIF category, yielding a field reader for each.
|
|
305
|
+
*
|
|
306
|
+
* The reader is a closure over the current row rather than an object, so
|
|
307
|
+
* nothing allocates a record per atom — this runs over hundreds of thousands of
|
|
308
|
+
* rows on a large entry, and the caller only ever wants a handful of the fields.
|
|
309
|
+
*
|
|
310
|
+
* Handles both forms a category takes: a `loop_` with a header block and many
|
|
311
|
+
* rows, and the flat `_category.field value` form mmCIF uses when there is
|
|
312
|
+
* exactly one row.
|
|
313
|
+
*/
|
|
314
|
+
function* mmcifRows(
|
|
315
|
+
lines: string[],
|
|
316
|
+
category: string
|
|
317
|
+
): Generator<(field: string) => string> {
|
|
318
|
+
const prefix = `${category}.`
|
|
319
|
+
|
|
320
|
+
for (let i = 0; i < lines.length; i++) {
|
|
321
|
+
const line = lines[i]!.trim()
|
|
322
|
+
|
|
323
|
+
// The flat, single-row form: consecutive `_category.field value` lines.
|
|
324
|
+
if (line.startsWith(prefix)) {
|
|
325
|
+
const single = new Map<string, string>()
|
|
326
|
+
while (i < lines.length) {
|
|
327
|
+
const item = lines[i]!.trim()
|
|
328
|
+
if (!item.startsWith(prefix)) break
|
|
329
|
+
const gap = item.search(/\s/)
|
|
330
|
+
if (gap === -1) {
|
|
331
|
+
// A field whose value is on the following line.
|
|
332
|
+
single.set(item.slice(prefix.length), unquote(lines[++i]?.trim() ?? ""))
|
|
333
|
+
} else {
|
|
334
|
+
single.set(
|
|
335
|
+
item.slice(prefix.length, gap),
|
|
336
|
+
unquote(item.slice(gap).trim())
|
|
337
|
+
)
|
|
338
|
+
}
|
|
339
|
+
i++
|
|
340
|
+
}
|
|
341
|
+
yield (field) => single.get(field) ?? ""
|
|
342
|
+
return
|
|
343
|
+
}
|
|
344
|
+
|
|
345
|
+
if (line !== "loop_") continue
|
|
346
|
+
|
|
347
|
+
// The header block: every `_category.field` line, in the order the values
|
|
348
|
+
// will arrive in.
|
|
349
|
+
const columns = new Map<string, number>()
|
|
350
|
+
let cursor = i + 1
|
|
351
|
+
while (cursor < lines.length) {
|
|
352
|
+
const header = lines[cursor]!.trim()
|
|
353
|
+
if (!header.startsWith("_")) break
|
|
354
|
+
if (header.startsWith(prefix)) {
|
|
355
|
+
columns.set(header.slice(prefix.length), columns.size)
|
|
356
|
+
}
|
|
357
|
+
cursor++
|
|
358
|
+
}
|
|
359
|
+
// A `loop_` for some other category. Its rows carry no leading underscore
|
|
360
|
+
// and aren't `loop_`, so the outer scan walks past them harmlessly.
|
|
361
|
+
if (columns.size === 0) continue
|
|
362
|
+
|
|
363
|
+
for (; cursor < lines.length; cursor++) {
|
|
364
|
+
const row = lines[cursor]!
|
|
365
|
+
const trimmed = row.trim()
|
|
366
|
+
// A loop ends at the next block, the next loop, or a blank line.
|
|
367
|
+
if (trimmed === "" || trimmed === "#" || trimmed.startsWith("_")) break
|
|
368
|
+
if (trimmed === "loop_" || trimmed.startsWith("data_")) break
|
|
369
|
+
|
|
370
|
+
const values = splitCifRow(trimmed)
|
|
371
|
+
yield (field) => {
|
|
372
|
+
const index = columns.get(field)
|
|
373
|
+
if (index === undefined) return ""
|
|
374
|
+
const value = values[index] ?? ""
|
|
375
|
+
return value === "." || value === "?" ? "" : value
|
|
376
|
+
}
|
|
377
|
+
}
|
|
378
|
+
return
|
|
379
|
+
}
|
|
380
|
+
}
|
|
381
|
+
|
|
382
|
+
/**
|
|
383
|
+
* Splits one mmCIF row into values, respecting quotes.
|
|
384
|
+
*
|
|
385
|
+
* Quoting is not decoration here: a chain identifier can be `'A'`, and a residue
|
|
386
|
+
* name can contain a space that a plain `split` would turn into two columns and
|
|
387
|
+
* shift every field after it by one.
|
|
388
|
+
*/
|
|
389
|
+
function splitCifRow(row: string): string[] {
|
|
390
|
+
const values: string[] = []
|
|
391
|
+
let i = 0
|
|
392
|
+
|
|
393
|
+
while (i < row.length) {
|
|
394
|
+
const char = row[i]!
|
|
395
|
+
if (char === " " || char === "\t") {
|
|
396
|
+
i++
|
|
397
|
+
continue
|
|
398
|
+
}
|
|
399
|
+
if (char === "'" || char === '"') {
|
|
400
|
+
const end = row.indexOf(char, i + 1)
|
|
401
|
+
if (end === -1) {
|
|
402
|
+
values.push(row.slice(i + 1))
|
|
403
|
+
break
|
|
404
|
+
}
|
|
405
|
+
values.push(row.slice(i + 1, end))
|
|
406
|
+
i = end + 1
|
|
407
|
+
continue
|
|
408
|
+
}
|
|
409
|
+
let end = i
|
|
410
|
+
while (end < row.length && row[end] !== " " && row[end] !== "\t") end++
|
|
411
|
+
values.push(row.slice(i, end))
|
|
412
|
+
i = end
|
|
413
|
+
}
|
|
414
|
+
|
|
415
|
+
return values
|
|
416
|
+
}
|
|
417
|
+
|
|
418
|
+
/** Strips the quotes mmCIF wraps a value in, and the null placeholders. */
|
|
419
|
+
function unquote(value: string): string {
|
|
420
|
+
const text = value.trim()
|
|
421
|
+
if (text === "." || text === "?") return ""
|
|
422
|
+
const first = text[0]
|
|
423
|
+
if ((first === "'" || first === '"') && text.endsWith(first) && text.length > 1) {
|
|
424
|
+
return text.slice(1, -1)
|
|
425
|
+
}
|
|
426
|
+
return text
|
|
427
|
+
}
|