asciichem 0.28.2 → 0.29.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  # frozen_string_literal: true
2
2
 
3
- require "parslet"
3
+ require 'parslet'
4
4
 
5
5
  module AsciiChem
6
6
  # Parslet grammar for AsciiChem v1.
@@ -17,461 +17,6 @@ module AsciiChem
17
17
  # ONLY the value, not the leading `_` or `^`. The literal prefix is
18
18
  # consumed by the rule but not included in the named capture.
19
19
  class Grammar < Parslet::Parser
20
- root :formula
21
-
22
- # -- top level ---------------------------------------------------------
23
-
24
- rule(:formula) do
25
- spaces? >> nodes.as(:formula) >> spaces?
26
- end
27
-
28
- rule(:nodes) { node >> (spaces? >> node).repeat }
29
-
30
- rule(:node) { reaction_cascade | reaction | electron_config | crystal | spectrum | calculation | zmatrix | mechanism | annotated_molecule | molecule | embedded_math | text_run.as(:text_run) }
31
-
32
- # -- crystallography -------------------------------------------------
33
-
34
- # crystal[Name](a=X,b=Y,...,sg=SG){atoms with @f(x,y,z)}
35
- rule(:crystal) do
36
- (str('crystal') >>
37
- crystal_name.maybe >>
38
- crystal_params.maybe >>
39
- crystal_body.maybe).as(:crystal_node)
40
- end
41
-
42
- rule(:crystal_name) do
43
- str('[') >> (str(']').absent? >> any).repeat.as(:crystal_name) >> str(']')
44
- end
45
-
46
- rule(:crystal_params) do
47
- str('(') >> (str(')').absent? >> any).repeat.as(:crystal_params) >> str(')')
48
- end
49
-
50
- rule(:crystal_body) do
51
- str('{') >> (str('}').absent? >> any).repeat.as(:crystal_body) >> str('}')
52
- end
53
-
54
- # -- spectroscopy ---------------------------------------------------
55
-
56
- # spectrum[type](params){peak data}
57
- rule(:spectrum) do
58
- (str('spectrum') >>
59
- spectrum_type.maybe >>
60
- spectrum_params.maybe >>
61
- spectrum_body.maybe).as(:spectrum_node)
62
- end
63
-
64
- rule(:spectrum_type) do
65
- str('[') >> (str(']').absent? >> any).repeat.as(:spectrum_type) >> str(']')
66
- end
67
-
68
- rule(:spectrum_params) do
69
- str('(') >> (str(')').absent? >> any).repeat.as(:spectrum_params) >> str(')')
70
- end
71
-
72
- rule(:spectrum_body) do
73
- str('{') >> (str('}').absent? >> any).repeat.as(:spectrum_body) >> str('}')
74
- end
75
-
76
- # -- computational chemistry ----------------------------------------
77
-
78
- # calc(method/basis){key: value units}
79
- rule(:calculation) do
80
- (str('calc') >>
81
- calc_params.maybe >>
82
- calc_body.maybe).as(:calc_node)
83
- end
84
-
85
- rule(:calc_params) do
86
- str('(') >> (str(')').absent? >> any).repeat.as(:calc_params) >> str(')')
87
- end
88
-
89
- rule(:calc_body) do
90
- str('{') >> (str('}').absent? >> any).repeat.as(:calc_body) >> str('}')
91
- end
92
-
93
- # -- Z-Matrix -------------------------------------------------------
94
-
95
- rule(:zmatrix) do
96
- (str('zmatrix') >>
97
- zmatrix_body.maybe).as(:zmatrix_node)
98
- end
99
-
100
- rule(:zmatrix_body) do
101
- str('{') >> (str('}').absent? >> any).repeat.as(:zmatrix_body) >> str('}')
102
- end
103
-
104
- # -- reaction mechanisms --------------------------------------------
105
-
106
- rule(:mechanism) do
107
- (str('mechanism') >>
108
- mechanism_body.maybe).as(:mechanism_node)
109
- end
110
-
111
- rule(:mechanism_body) do
112
- str('{') >> (str('}').absent? >> any).repeat.as(:mechanism_body) >> str('}')
113
- end
114
-
115
- # Annotated molecule: a molecule followed by one or more
116
- # `@key("value")` annotations for CML metadata (names,
117
- # identifiers, title, formula, labels).
118
- rule(:annotated_molecule) do
119
- molecule.as(:mol) >> molecule_annotation.repeat(1).as(:annotations)
120
- end
121
-
122
- rule(:molecule_annotation) do
123
- spaces?.maybe >>
124
- (metadata_annotation | simple_annotation)
125
- end
126
-
127
- # Metadata: @meta("key","value") — two quoted args, comma-separated.
128
- # Produces CML <metadata name="key" content="value"/>.
129
- # Uses distinct capture keys (meta_key/meta_value) so the transform
130
- # can distinguish metadata from regular @key("value") annotations.
131
- rule(:metadata_annotation) do
132
- str('@meta(') >> str('"') >>
133
- (str('"').absent? >> any).repeat.as(:meta_key) >> str('"') >>
134
- str(',') >> str('"') >>
135
- (str('"').absent? >> any).repeat.as(:meta_value) >> str('"') >>
136
- str(')')
137
- end
138
-
139
- # Simple annotation: @key("value") — one quoted arg.
140
- # Known types (name, inchi, etc.) are handled specially by the
141
- # transform. Unknown types become properties.
142
- rule(:simple_annotation) do
143
- str('@') >> annotation_type.as(:ann_type) >>
144
- str('(') >> str('"') >>
145
- (str('"').absent? >> any).repeat.as(:ann_value) >>
146
- str('"') >> str(')')
147
- end
148
-
149
- # Known annotation types first; property_name is a catch-all so
150
- # any lowercase word (e.g. "mw", "density", "logP") becomes a
151
- # property annotation.
152
- rule(:annotation_type) do
153
- str('name') | str('title') | str('formula') | str('label') |
154
- str('inchi') | str('smiles') | str('cas') | str('iupac') |
155
- str('cid') | str('chebi') |
156
- property_name
157
- end
158
-
159
- rule(:property_name) do
160
- match('[a-z]').repeat(1)
161
- end
162
-
163
- # -- reactions ---------------------------------------------------------
164
-
165
- # A reaction cascade is two or more reactions chained together:
166
- # the products of step N become the reactants of step N+1, with
167
- # an arrow between them. The grammar reuses `reaction` for the
168
- # first leg and `arrow >> terms` for each subsequent leg; the
169
- # transform promotes the whole chain to a `ReactionCascade`.
170
- rule(:reaction_cascade) do
171
- (reaction.as(:first) >>
172
- (arrow.as(:arrow) >> spaces? >> terms.as(:products)).repeat(1)).as(:cascade)
173
- end
174
-
175
- rule(:reaction) do
176
- terms.as(:reactants) >>
177
- arrow.as(:arrow) >>
178
- spaces? >>
179
- terms.as(:products)
180
- end
181
-
182
- rule(:terms) do
183
- molecule >> (spaces? >> plus >> spaces? >> molecule).repeat
184
- end
185
-
186
- rule(:plus) { str("+") }
187
-
188
- rule(:arrow) do
189
- spaces? >>
190
- arrow_token.as(:kind) >>
191
- condition.maybe.as(:above) >>
192
- condition.maybe.as(:below)
193
- end
194
-
195
- rule(:condition) do
196
- str("[") >> (str("]").absent? >> any).repeat.as(:text) >> str("]")
197
- end
198
-
199
- rule(:arrow_token) do
200
- str("<=>") | str("<->") | str("->") | str("<-")
201
- end
202
-
203
- # -- molecules ---------------------------------------------------------
204
-
205
- rule(:molecule) do
206
- stereo_prefix.maybe >> coefficient.maybe >> units.as(:units)
207
- end
208
-
209
- # Stereochemistry prefix: `(R)-`, `(S)-`, `(E)-`, `(Z)-`,
210
- # `(a)-`/`(α)-` (alpha), `(b)-`/`(β)-` (beta). Tried before
211
- # `coefficient` so the lookahead-via-failure on the closed letter
212
- # set disambiguates from a parenthesised group: `(R)` matches
213
- # because `R` is in the stereo set; `(OH)` fails because `OH` is
214
- # not a single stereo letter, and the molecule rule falls through
215
- # to the regular group parse.
216
- rule(:stereo_prefix) do
217
- str("(") >> stereo_letter.as(:stereo) >> str(")") >> str("-")
218
- end
219
-
220
- rule(:stereo_letter) do
221
- str("alpha") | str("beta") |
222
- str("R") | str("S") | str("E") | str("Z") |
223
- str("α") | str("β") |
224
- str("a") | str("b")
225
- end
226
-
227
- rule(:units) { (unit | bond).repeat(1) }
228
- rule(:unit) { prefixed_atom | group | hydrogen_atom | plain_atom }
229
-
230
- # Hydrogen special case: H cannot form ring closures (only 1 bond),
231
- # so bare digits after H are unambiguously subscripts. This lets
232
- # users write `H2O` instead of `H_2O` for the common formula case
233
- # without conflicting with SMILES-style ring closures on other
234
- # elements (`C1-C-C1` is still cyclohexane-style).
235
- #
236
- # The `match('[a-z]').absent?` lookahead prevents this rule from
237
- # stealing the H from `He`, `Ho`, etc. — those fall through to
238
- # plain_atom which matches the full element symbol greedily.
239
- rule(:hydrogen_atom) do
240
- (lewis_prefix.maybe >>
241
- str('H').as(:element) >>
242
- match('[a-z]').absent? >>
243
- h_subscript.maybe.as(:subscript) >>
244
- superscript_marker.maybe.as(:superscript) >>
245
- lewis_radicals.maybe >>
246
- atom_annotation).as(:atom)
247
- end
248
-
249
- # Subscript on hydrogen: explicit `_digits` OR bare `digits`
250
- # (implicit). Both produce the same `:subscript` capture. Bare
251
- # digits are safe here because H can't ring-close.
252
- rule(:h_subscript) do
253
- (str('_') >> subscript_value) | digits
254
- end
255
-
256
- # Bonds appear inside molecules as separators between units.
257
- # Supported kinds, in alternation order (longest match first to
258
- # avoid `>-` shadowing `-`):
259
- # single `-`
260
- # double `=`
261
- # triple `#`
262
- # quadruple `##`
263
- # wedge `>-` (solid wedge toward viewer)
264
- # hash `-<` (hashed wedge away from viewer)
265
- # dative `~>` (electron-pair donor → acceptor; `->` is taken
266
- # by the reaction arrow)
267
- # wavy `~~` (resonance / delocalised)
268
- rule(:bond) do
269
- str("##").as(:quadruple) |
270
- str(">-").as(:wedge) |
271
- str("-<").as(:hash) |
272
- str("~>").as(:dative) |
273
- str("~~").as(:wavy) |
274
- str("#").as(:triple) |
275
- str("=").as(:double) |
276
- str("-").as(:single)
277
- end
278
-
279
- rule(:coefficient) do
280
- (digits.as(:value) >> (element_symbol | open_bracket).present?).as(:coefficient)
281
- end
282
-
283
- rule(:open_bracket) { str("(") | str("[") | str("{") }
284
-
285
- # -- atoms -------------------------------------------------------------
286
-
287
- rule(:prefixed_atom) do
288
- (lewis_prefix.maybe >>
289
- isotope_marker.as(:isotope) >>
290
- element_symbol.as(:element) >>
291
- atom_suffix >>
292
- lewis_radicals.maybe >>
293
- ring_closures.maybe.as(:ring_closures) >>
294
- atom_annotation).as(:atom)
295
- end
296
-
297
- rule(:plain_atom) do
298
- (lewis_prefix.maybe >>
299
- element_symbol.as(:element) >>
300
- atom_suffix >>
301
- lewis_radicals.maybe >>
302
- ring_closures.maybe.as(:ring_closures) >>
303
- atom_annotation).as(:atom)
304
- end
305
-
306
- # Atom annotations: each is independently optional via .maybe.
307
- # Order matters: @(x,y) → @R/@S → @m(N) → @t("...") → @f(x,y,z).
308
- # Example: C@(10,20)@R@m(2)@t("C1")@f(0.5,0.5,0.5)
309
- rule(:atom_annotation) do
310
- coordinate_annotation.maybe >>
311
- parity_annotation.maybe >>
312
- multiplicity_annotation.maybe >>
313
- atom_title_annotation.maybe >>
314
- fractional_annotation.maybe
315
- end
316
-
317
- rule(:parity_annotation) do
318
- str('@') >> (str('R') | str('S')).as(:atom_parity)
319
- end
320
-
321
- rule(:coordinate_annotation) do
322
- str('@(') >>
323
- float_number.as(:x2) >> str(',') >>
324
- float_number.as(:y2) >>
325
- (str(',') >> float_number.as(:z2)).maybe >>
326
- str(')')
327
- end
328
-
329
- # Spin multiplicity: @m(2) for doublet, @m(1) for singlet
330
- rule(:multiplicity_annotation) do
331
- str('@m(') >> match('[0-9]').repeat(1).as(:spin_multiplicity) >> str(')')
332
- end
333
-
334
- # Atom title/label: @t("C1")
335
- rule(:atom_title_annotation) do
336
- str('@t(') >> str('"') >>
337
- (str('"').absent? >> any).repeat.as(:atom_title) >>
338
- str('"') >> str(')')
339
- end
340
-
341
- # Fractional coordinates (crystallographic): @f(0.5,0.25,0.75)
342
- rule(:fractional_annotation) do
343
- str('@f(') >>
344
- float_number.as(:x_fract) >> str(',') >>
345
- float_number.as(:y_fract) >> str(',') >>
346
- float_number.as(:z_fract) >>
347
- str(')')
348
- end
349
-
350
- rule(:float_number) do
351
- str('-').maybe >> match('[0-9]').repeat(1) >> (str('.') >> match('[0-9]').repeat(0)).maybe
352
- end
353
-
354
- # Ring closure digits (SMILES-style). A digit suffix on an atom
355
- # opens or closes a ring; two atoms with the same digit become
356
- # bonded. Captured as a string (e.g. "1", "12") so multiple
357
- # closures on one atom are preserved.
358
- rule(:ring_closures) do
359
- match('[0-9]').repeat(1)
360
- end
361
-
362
- rule(:atom_suffix) do
363
- subscript_marker.maybe.as(:subscript) >>
364
- superscript_marker.maybe.as(:superscript)
365
- end
366
-
367
- # Lewis markers. Prefix `:` count = lone_pairs (binds to following
368
- # atom, like the isotope prefix). Suffix `.` count = radical
369
- # electrons. Position-specific lone pairs (`:O:` style) collapse
370
- # into a total count — the renderer decides layout.
371
- rule(:lewis_prefix) { str(":").repeat(1).as(:lone_pairs) }
372
- rule(:lewis_radicals) { str(".").repeat(1).as(:radical_electrons) }
373
-
374
- # Markers consume the leading `_` / `^` but capture only the value.
375
- rule(:isotope_marker) do
376
- (str("^") | str("_")) >> digits
377
- end
378
-
379
- rule(:subscript_marker) do
380
- str("_") >> subscript_value
381
- end
382
-
383
- rule(:superscript_marker) do
384
- str("^") >> superscript_value
385
- end
386
-
387
- rule(:element_symbol) do
388
- match("[A-Z]") >> match("[a-z]").maybe
389
- end
390
-
391
- rule(:subscript_value) do
392
- (str("{") >> (str("}").absent? >> any).repeat >> str("}")) |
393
- digits
394
- end
395
-
396
- rule(:superscript_value) do
397
- oxidation_state | charge | braced_or_bare
398
- end
399
-
400
- rule(:charge) do
401
- (digits >> match("[+-]")) |
402
- (match("[+-]") >> digits.maybe) |
403
- digits
404
- end
405
-
406
- rule(:oxidation_state) do
407
- str("(") >> roman_numeral >> str(")")
408
- end
409
-
410
- rule(:roman_numeral) { match("[IVXLCDM]").repeat(1) }
411
-
412
- rule(:braced_or_bare) do
413
- (str("{") >> (str("}").absent? >> any).repeat >> str("}")) |
414
- match("[0-9a-zA-Z]").repeat(1)
415
- end
416
-
417
- # -- groups ------------------------------------------------------------
418
-
419
- rule(:group) do
420
- group_open.as(:open_bracket) >>
421
- group_nodes >>
422
- group_close.as(:close_bracket) >>
423
- multiplicity.maybe.as(:multiplicity)
424
- end
425
-
426
- # `group_nodes` is a separate rule so we can carve out the closing
427
- # bracket from `node`'s text fallback. Without this, `text_run`
428
- # would consume the closing bracket and the group never terminates.
429
- rule(:group_nodes) do
430
- group_node.repeat(1).as(:group_nodes)
431
- end
432
-
433
- rule(:group_node) do
434
- reaction | electron_config | molecule | embedded_math | group_text_run
435
- end
436
-
437
- rule(:group_text_run) do
438
- str('"') >> (str('"').absent? >> any).repeat >> str('"')
439
- end
440
-
441
- rule(:group_open) { str("(") | str("[") | str("{") }
442
- rule(:group_close) { str(")") | str("]") | str("}") }
443
-
444
- rule(:multiplicity) { str("_") >> digits }
445
-
446
- # -- electron configuration -------------------------------------------
447
-
448
- rule(:electron_config) do
449
- (orbital.as(:orbital) >> str("^") >> digits.as(:occupancy) >> spaces?.maybe).repeat(2).as(:electron_config)
450
- end
451
-
452
- rule(:orbital) { digits >> match("[spdfgh]") }
453
-
454
- # -- embedded math ----------------------------------------------------
455
-
456
- rule(:embedded_math) do
457
- str("`") >>
458
- (str("`").absent? >> any).repeat.as(:math_source) >>
459
- str("`")
460
- end
461
-
462
- # -- text (top level) -------------------------------------------------
463
-
464
- # Free-form text uses `"..."` delimiters, matching AsciiMath's
465
- # convention. The quoted content becomes a Text node with the
466
- # surrounding quotes stripped (handled by the transform).
467
- rule(:text_run) do
468
- str('"') >> (str('"').absent? >> any).repeat >> str('"')
469
- end
470
-
471
- # -- primitives -------------------------------------------------------
472
-
473
- rule(:digits) { match("[0-9]").repeat(1) }
474
- rule(:spaces) { match(/\s/).repeat(1) }
475
- rule(:spaces?) { spaces.maybe }
20
+ include GrammarRules
476
21
  end
477
22
  end