asciichem 0.20.0 → 0.21.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +25 -1
- data/lib/asciichem/formatter/structural_svg.rb +8 -0
- data/lib/asciichem/model/atom.rb +8 -3
- data/lib/asciichem/model/bond.rb +2 -1
- data/lib/asciichem/model/node.rb +12 -0
- data/lib/asciichem/molfile/parser.rb +169 -0
- data/lib/asciichem/molfile/writer.rb +91 -0
- data/lib/asciichem/molfile.rb +26 -0
- data/lib/asciichem/smiles/parser.rb +302 -0
- data/lib/asciichem/smiles/writer.rb +210 -0
- data/lib/asciichem/smiles.rb +45 -0
- data/lib/asciichem/structure/graph.rb +62 -0
- data/lib/asciichem/structure/linearizer.rb +67 -0
- data/lib/asciichem/structure.rb +20 -0
- data/lib/asciichem/version.rb +1 -1
- data/lib/asciichem/wire/core.rb +4 -0
- data/lib/asciichem/wire_adapter.rb +9 -2
- data/lib/asciichem.rb +16 -0
- metadata +10 -1
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: 2dee0024c277879fed3b5695a76e336067660467724663c4a67f71ec87215e80
|
|
4
|
+
data.tar.gz: 50d5a53fd8d4f64a0d3969c586e81552faed9221a8f209932cd6823a2a966327
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: 3afdd1db8e99729ca9a1a49be7a748fd56d1fdb666b5bfe10bb03a08c6d7db440ca82b3c900b6f22f09ac66901d7c086913a2a3e8c142bc6c7d776f21959c629
|
|
7
|
+
data.tar.gz: ba3091db4ea3a3d7ba8dcd28f9689df323b135e1e1e4f0a3f3573f14350448461f9f3ade4e20fe4e7af9408d0ce40c1764bba4d609a05bd770705be431ca497b
|
data/CHANGELOG.md
CHANGED
|
@@ -3,6 +3,29 @@
|
|
|
3
3
|
All notable changes to AsciiChem are documented here.
|
|
4
4
|
This project follows [Semantic Versioning](https://semver.org/).
|
|
5
5
|
|
|
6
|
+
## [0.21.0] - 2026-09-12
|
|
7
|
+
|
|
8
|
+
### Added
|
|
9
|
+
- Structure interchange (TODO.v2 09, TODO.impl 57): SMILES and
|
|
10
|
+
molfile (CTfile V2000) ingestion and emission as modules of the one
|
|
11
|
+
semantic model — `AsciiChem.parse_smiles` / `parse_molfile`,
|
|
12
|
+
`to_smiles` / `to_molfile`. Ingested molecules are ordinary
|
|
13
|
+
`Model::Molecule`s: graphs linearise into atoms + bond tokens +
|
|
14
|
+
ring-closure digits (`Structure::Linearizer`), so every existing
|
|
15
|
+
renderer, linter, and wire form works unchanged. The SMILES writer
|
|
16
|
+
is deterministic (DFS, single-bond continuations, order-independent
|
|
17
|
+
tie-breaks); aspirin and naphthalene round-trip exactly. v1
|
|
18
|
+
deferrals, each with an actionable `ParseError`: chirality `@`/`@@`,
|
|
19
|
+
E/Z directions `/` `\`, wildcard atoms, bonded ring closures.
|
|
20
|
+
- `Model::Atom#aromatic` / `#hydrogens` and an `aromatic` bond kind
|
|
21
|
+
(asciichem-model 0.4.0 fields): lowercase SMILES atoms, bracket
|
|
22
|
+
H-counts, molfile type-4 bonds (aromatic atoms marked from bonds).
|
|
23
|
+
- `AsciiChem::Structure` — shared graph walk + adjacency linearizer
|
|
24
|
+
for the interchange formats; `StructuralSvg` renders aromatic bonds
|
|
25
|
+
dashed; wire form carries the new fields both ways.
|
|
26
|
+
- Corpus levels: asciichem-tests v0.3.0 `structure/smiles/*` and
|
|
27
|
+
`structure/molfile/*` fixtures at 100% (37 + 6 cases).
|
|
28
|
+
|
|
6
29
|
## [0.20.0] - 2026-09-12
|
|
7
30
|
|
|
8
31
|
### Added
|
|
@@ -363,7 +386,8 @@ This project follows [Semantic Versioning](https://semver.org/).
|
|
|
363
386
|
`version`.
|
|
364
387
|
- Comprehensive RSpec suite with round-trip conformance.
|
|
365
388
|
|
|
366
|
-
[Unreleased]: https://github.com/asciichem/asciichem-ruby/compare/v0.
|
|
389
|
+
[Unreleased]: https://github.com/asciichem/asciichem-ruby/compare/v0.21.0...HEAD
|
|
390
|
+
[0.21.0]: https://github.com/asciichem/asciichem-ruby/compare/v0.20.0...v0.21.0
|
|
367
391
|
[0.20.0]: https://github.com/asciichem/asciichem-ruby/compare/v0.19.0...v0.20.0
|
|
368
392
|
[0.18.1]: https://github.com/asciichem/asciichem-ruby/compare/v0.18.0...v0.18.1
|
|
369
393
|
[0.18.0]: https://github.com/asciichem/asciichem-ruby/compare/v0.17.0...v0.18.0
|
|
@@ -258,6 +258,14 @@ module AsciiChem
|
|
|
258
258
|
RENDERERS = {
|
|
259
259
|
single: ->(r) { [r.base_line] },
|
|
260
260
|
|
|
261
|
+
# Aromatic bonds render as dashed lines (the inner-ring
|
|
262
|
+
# circle is a renderer nicety left for a later iteration).
|
|
263
|
+
aromatic: lambda do |r|
|
|
264
|
+
line = r.base_line
|
|
265
|
+
line['stroke-dasharray'] = '4 2.5'
|
|
266
|
+
[line]
|
|
267
|
+
end,
|
|
268
|
+
|
|
261
269
|
double: ->(r) { [-SPACING, 0, SPACING].map { |d| r.offset_line(d) }.compact },
|
|
262
270
|
|
|
263
271
|
triple: ->(r) { [0, -SPACING * 1.5, SPACING * 1.5].map { |d| r.offset_line(d) }.compact },
|
data/lib/asciichem/model/atom.rb
CHANGED
|
@@ -34,7 +34,7 @@ module AsciiChem
|
|
|
34
34
|
attr_accessor :element, :isotope, :subscript, :superscript,
|
|
35
35
|
:charge, :oxidation_state,
|
|
36
36
|
:lone_pairs, :radical_electrons,
|
|
37
|
-
:ring_closures,
|
|
37
|
+
:ring_closures, :aromatic, :hydrogens,
|
|
38
38
|
:x2, :y2, :z2, :atom_parity,
|
|
39
39
|
:spin_multiplicity, :atom_title,
|
|
40
40
|
:x_fract, :y_fract, :z_fract
|
|
@@ -58,7 +58,7 @@ module AsciiChem
|
|
|
58
58
|
def initialize(element:, isotope: nil, subscript: nil,
|
|
59
59
|
superscript: nil, charge: nil, oxidation_state: nil,
|
|
60
60
|
lone_pairs: nil, radical_electrons: nil,
|
|
61
|
-
ring_closures: nil,
|
|
61
|
+
ring_closures: nil, aromatic: nil, hydrogens: nil,
|
|
62
62
|
x2: nil, y2: nil, z2: nil, atom_parity: nil,
|
|
63
63
|
spin_multiplicity: nil, atom_title: nil,
|
|
64
64
|
x_fract: nil, y_fract: nil, z_fract: nil)
|
|
@@ -71,6 +71,8 @@ module AsciiChem
|
|
|
71
71
|
@lone_pairs = lone_pairs
|
|
72
72
|
@radical_electrons = radical_electrons
|
|
73
73
|
@ring_closures = ring_closures
|
|
74
|
+
@aromatic = aromatic
|
|
75
|
+
@hydrogens = hydrogens
|
|
74
76
|
@x2 = x2
|
|
75
77
|
@y2 = y2
|
|
76
78
|
@z2 = z2
|
|
@@ -87,7 +89,8 @@ module AsciiChem
|
|
|
87
89
|
superscript: superscript, charge: charge,
|
|
88
90
|
oxidation_state: oxidation_state,
|
|
89
91
|
lone_pairs: lone_pairs, radical_electrons: radical_electrons,
|
|
90
|
-
ring_closures: ring_closures,
|
|
92
|
+
ring_closures: ring_closures, aromatic: aromatic,
|
|
93
|
+
hydrogens: hydrogens,
|
|
91
94
|
x2: x2, y2: y2, z2: z2, atom_parity: atom_parity,
|
|
92
95
|
spin_multiplicity: spin_multiplicity, atom_title: atom_title,
|
|
93
96
|
x_fract: x_fract, y_fract: y_fract, z_fract: z_fract }
|
|
@@ -108,6 +111,8 @@ module AsciiChem
|
|
|
108
111
|
parts << "^(#{oxidation_state})" if oxidation_state
|
|
109
112
|
parts << ".#{radical_electrons}" if radical_electrons
|
|
110
113
|
parts << ring_closures.to_s if ring_closures
|
|
114
|
+
parts << "aromatic" if aromatic
|
|
115
|
+
parts << "H#{hydrogens}" if hydrogens
|
|
111
116
|
"Atom(#{parts.join})"
|
|
112
117
|
end
|
|
113
118
|
end
|
data/lib/asciichem/model/bond.rb
CHANGED
|
@@ -14,7 +14,8 @@ module AsciiChem
|
|
|
14
14
|
wedge: { ascii: ">-", mathml_entity: "↑" },
|
|
15
15
|
hash: { ascii: "-<", mathml_entity: "↓" },
|
|
16
16
|
dative: { ascii: "~>", mathml_entity: "→" },
|
|
17
|
-
wavy: { ascii: "~~", mathml_entity: "∼" }
|
|
17
|
+
wavy: { ascii: "~~", mathml_entity: "∼" },
|
|
18
|
+
aromatic: { ascii: ":", mathml_entity: ":" }
|
|
18
19
|
}.freeze
|
|
19
20
|
|
|
20
21
|
# CML wire order codes per bond kind. Single source of truth
|
data/lib/asciichem/model/node.rb
CHANGED
|
@@ -70,6 +70,18 @@ module AsciiChem
|
|
|
70
70
|
AsciiChem::WireAdapter.to_model_json(self)
|
|
71
71
|
end
|
|
72
72
|
|
|
73
|
+
# Deterministic SMILES for a Molecule (or dot-joined components
|
|
74
|
+
# for a Formula). Raises for constructs with no SMILES form.
|
|
75
|
+
def to_smiles
|
|
76
|
+
AsciiChem::Smiles.write(self)
|
|
77
|
+
end
|
|
78
|
+
|
|
79
|
+
# Molfile V2000 for a Molecule. Authored coordinates win; a
|
|
80
|
+
# deterministic 2D layout is computed otherwise.
|
|
81
|
+
def to_molfile(name: nil)
|
|
82
|
+
AsciiChem::Molfile.write(self, name: name)
|
|
83
|
+
end
|
|
84
|
+
|
|
73
85
|
# Subclasses override to expose the attributes that participate in
|
|
74
86
|
# equality. Default: empty (so two bare Nodes are equal).
|
|
75
87
|
def value_attributes
|
|
@@ -0,0 +1,169 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module AsciiChem
|
|
4
|
+
module Molfile
|
|
5
|
+
# V2000 molfile → Model::Molecule. Fixed-format blocks: header
|
|
6
|
+
# (3 lines), counts line, atom block, bond block, property block.
|
|
7
|
+
# Charges come from `M CHG`, isotopes from the mass-difference
|
|
8
|
+
# field or `M ISO`, bond stereo codes 1/6 become wedge/hash.
|
|
9
|
+
class Parser
|
|
10
|
+
def initialize(text)
|
|
11
|
+
@lines = text.lines.map(&:chomp)
|
|
12
|
+
end
|
|
13
|
+
|
|
14
|
+
def parse
|
|
15
|
+
raise ParseError, "molfile too short" if @lines.length < 5
|
|
16
|
+
|
|
17
|
+
atom_field = field(3, 0, 3)
|
|
18
|
+
bond_field = field(3, 3, 3)
|
|
19
|
+
unless atom_field.match?(/\A\d+\z/) && bond_field.match?(/\A\d+\z/)
|
|
20
|
+
raise ParseError, "malformed counts line: #{@lines[3].inspect}"
|
|
21
|
+
end
|
|
22
|
+
atom_count = atom_field.to_i
|
|
23
|
+
bond_count = bond_field.to_i
|
|
24
|
+
|
|
25
|
+
atoms = parse_atoms(atom_count)
|
|
26
|
+
bonds = parse_bonds(bond_count)
|
|
27
|
+
properties = parse_properties
|
|
28
|
+
|
|
29
|
+
apply_legacy_charge!(atoms)
|
|
30
|
+
apply_charges!(atoms, properties[:charges])
|
|
31
|
+
apply_isotopes!(atoms, properties[:isotopes])
|
|
32
|
+
|
|
33
|
+
adjacency = Array.new(atoms.length) { {} }
|
|
34
|
+
bonds.each do |bond|
|
|
35
|
+
adjacency[bond[:from]][bond[:to]] = bond[:kind]
|
|
36
|
+
adjacency[bond[:to]][bond[:from]] = bond[:kind]
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
# V2000 carries aromaticity on bonds (type 4); the model
|
|
40
|
+
# carries it on atoms and bonds, so atoms touching an
|
|
41
|
+
# aromatic bond are marked aromatic.
|
|
42
|
+
bonds.each do |bond|
|
|
43
|
+
next unless bond[:kind] == :aromatic
|
|
44
|
+
|
|
45
|
+
atoms[bond[:from]].aromatic = true
|
|
46
|
+
atoms[bond[:to]].aromatic = true
|
|
47
|
+
end
|
|
48
|
+
|
|
49
|
+
Model::Molecule.new(
|
|
50
|
+
nodes: Structure::Linearizer.new(atoms: atoms, edges: adjacency_to_edges(adjacency)).nodes
|
|
51
|
+
)
|
|
52
|
+
end
|
|
53
|
+
|
|
54
|
+
private
|
|
55
|
+
|
|
56
|
+
def field(line_index, start, length)
|
|
57
|
+
(@lines[line_index] || "")[start, length].to_s.strip
|
|
58
|
+
end
|
|
59
|
+
|
|
60
|
+
def parse_atoms(count)
|
|
61
|
+
(1..count).map do |i|
|
|
62
|
+
line_index = 3 + i
|
|
63
|
+
line = @lines[line_index]
|
|
64
|
+
raise ParseError, "truncated atom block (expected #{count} atoms)" if line.nil?
|
|
65
|
+
|
|
66
|
+
x = line[0, 10].to_f
|
|
67
|
+
y = line[10, 10].to_f
|
|
68
|
+
z = line[20, 10].to_f
|
|
69
|
+
element = line[31, 3].to_s.strip
|
|
70
|
+
mass_diff = line[34, 2].to_i
|
|
71
|
+
raise ParseError, "atom #{i} has no element symbol" if element.empty?
|
|
72
|
+
|
|
73
|
+
Model::Atom.new(
|
|
74
|
+
element: element,
|
|
75
|
+
x2: x, y2: y, z2: z,
|
|
76
|
+
isotope: isotope_from_mass_diff(element, mass_diff)
|
|
77
|
+
)
|
|
78
|
+
end
|
|
79
|
+
end
|
|
80
|
+
|
|
81
|
+
def parse_bonds(count)
|
|
82
|
+
(1..count).map do |i|
|
|
83
|
+
line = @lines[3 + atoms_count + i]
|
|
84
|
+
raise ParseError, "truncated bond block (expected #{count} bonds)" if line.nil?
|
|
85
|
+
|
|
86
|
+
from = line[0, 3].to_i - 1
|
|
87
|
+
to = line[3, 3].to_i - 1
|
|
88
|
+
type = line[6, 3].to_i
|
|
89
|
+
stereo = line[9, 3].to_i
|
|
90
|
+
raise ParseError, "bond #{i} has out-of-range atom indexes" if from.negative? || to.negative?
|
|
91
|
+
|
|
92
|
+
{ from: from, to: to, kind: bond_kind(type, stereo) }
|
|
93
|
+
end
|
|
94
|
+
end
|
|
95
|
+
|
|
96
|
+
def parse_properties
|
|
97
|
+
charges = {}
|
|
98
|
+
isotopes = {}
|
|
99
|
+
@lines.each do |line|
|
|
100
|
+
if line.start_with?("M CHG")
|
|
101
|
+
parts = line[6..].split
|
|
102
|
+
_count = parts[0].to_i
|
|
103
|
+
parts[1..].each_slice(2) do |idx, charge|
|
|
104
|
+
charges[idx.to_i - 1] = charge.to_i if idx && charge
|
|
105
|
+
end
|
|
106
|
+
elsif line.start_with?("M ISO")
|
|
107
|
+
parts = line[6..].split
|
|
108
|
+
parts[1..].each_slice(2) do |idx, mass|
|
|
109
|
+
isotopes[idx.to_i - 1] = mass.to_i if idx && mass
|
|
110
|
+
end
|
|
111
|
+
end
|
|
112
|
+
end
|
|
113
|
+
{ charges: charges, isotopes: isotopes }
|
|
114
|
+
end
|
|
115
|
+
|
|
116
|
+
def bond_kind(type, stereo)
|
|
117
|
+
return :wedge if stereo == 1
|
|
118
|
+
return :hash if stereo == 6
|
|
119
|
+
|
|
120
|
+
{ 1 => :single, 2 => :double, 3 => :triple, 4 => :aromatic }[type] ||
|
|
121
|
+
raise(ParseError, "unsupported molfile bond type #{type}")
|
|
122
|
+
end
|
|
123
|
+
|
|
124
|
+
# Pre-CHG charge column (0-based 36, 3): nonzero values 1..4
|
|
125
|
+
# mean +1..+4, 5..7 mean -1..-3. `M CHG` overrides.
|
|
126
|
+
def apply_legacy_charge!(atoms)
|
|
127
|
+
(1..field(3, 0, 3).to_i).each do |i|
|
|
128
|
+
code = field(3 + i, 36, 3).to_i
|
|
129
|
+
next if code.zero?
|
|
130
|
+
|
|
131
|
+
charge = code <= 4 ? code : 4 - code
|
|
132
|
+
sign = charge.negative? ? "-" : "+"
|
|
133
|
+
atoms[i - 1].charge = charge.abs == 1 ? sign : "#{charge.abs}#{sign}"
|
|
134
|
+
end
|
|
135
|
+
end
|
|
136
|
+
|
|
137
|
+
def apply_charges!(atoms, charges)
|
|
138
|
+
charges.each do |index, value|
|
|
139
|
+
sign = value.negative? ? "-" : "+"
|
|
140
|
+
atoms[index].charge = value.abs == 1 ? sign : "#{value.abs}#{sign}"
|
|
141
|
+
end
|
|
142
|
+
end
|
|
143
|
+
|
|
144
|
+
def apply_isotopes!(atoms, isotopes)
|
|
145
|
+
isotopes.each do |index, mass|
|
|
146
|
+
atoms[index].isotope = mass.to_s
|
|
147
|
+
end
|
|
148
|
+
end
|
|
149
|
+
|
|
150
|
+
# Mass difference encodes isotopes relative to the rounded
|
|
151
|
+
# average mass; mapping it unambiguously requires isotope
|
|
152
|
+
# tables, so v1 defers isotopes to the explicit `M ISO` block.
|
|
153
|
+
def isotope_from_mass_diff(_element, mass_diff)
|
|
154
|
+
nil
|
|
155
|
+
end
|
|
156
|
+
def atoms_count
|
|
157
|
+
field(3, 0, 3).to_i
|
|
158
|
+
end
|
|
159
|
+
|
|
160
|
+
def adjacency_to_edges(adjacency)
|
|
161
|
+
edges = []
|
|
162
|
+
adjacency.each_with_index do |neighbors, index|
|
|
163
|
+
neighbors.each { |to, kind| edges << Structure::Graph::Edge.new(from: index, to: to, kind: kind) if to > index }
|
|
164
|
+
end
|
|
165
|
+
edges
|
|
166
|
+
end
|
|
167
|
+
end
|
|
168
|
+
end
|
|
169
|
+
end
|
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module AsciiChem
|
|
4
|
+
module Molfile
|
|
5
|
+
# Model molecule → V2000 molfile. Authored x2/y2 coordinates are
|
|
6
|
+
# used when present; otherwise a deterministic 2D layout is
|
|
7
|
+
# computed (Layout walks atoms in the same order as
|
|
8
|
+
# Structure::Graph, so positions map by index). Charges and
|
|
9
|
+
# isotopes are emitted as `M CHG` / `M ISO` property lines.
|
|
10
|
+
class Writer
|
|
11
|
+
BOND_TYPES = {
|
|
12
|
+
single: 1, double: 2, triple: 3, aromatic: 4,
|
|
13
|
+
wedge: 1, hash: 1
|
|
14
|
+
}.freeze
|
|
15
|
+
BOND_STEREO = { wedge: 1, hash: 6 }.freeze
|
|
16
|
+
|
|
17
|
+
def initialize(molecule, name: nil)
|
|
18
|
+
@molecule = molecule
|
|
19
|
+
@name = name
|
|
20
|
+
end
|
|
21
|
+
|
|
22
|
+
def write
|
|
23
|
+
atoms, edges = Structure::Graph.build(@molecule)
|
|
24
|
+
raise ParseError, "molecule has no bonds — a formula is not a structure" if edges.empty?
|
|
25
|
+
|
|
26
|
+
lines = []
|
|
27
|
+
lines << @name.to_s
|
|
28
|
+
lines << " AsciiChem"
|
|
29
|
+
lines << ""
|
|
30
|
+
lines << format("%3d%3d 0 0 0 0 0 0 0 0999 V2000", atoms.length, edges.length)
|
|
31
|
+
|
|
32
|
+
layout = nil
|
|
33
|
+
atoms.each_with_index do |atom, index|
|
|
34
|
+
x, y = coordinates(atom, index, atoms)
|
|
35
|
+
lines << format("%10.4f%10.4f%10.4f %-3s 0 0 0 0 0 0 0 0 0 0 0 0",
|
|
36
|
+
x, y, atom.z2 || 0.0, atom.element)
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
edges.each do |edge|
|
|
40
|
+
type = BOND_TYPES.fetch(edge.kind) do
|
|
41
|
+
raise ParseError, "#{edge.kind} bonds have no molfile V2000 type"
|
|
42
|
+
end
|
|
43
|
+
stereo = BOND_STEREO.fetch(edge.kind, 0)
|
|
44
|
+
lines << format("%3d%3d%3d%3d 0 0 0 0 0 0 0",
|
|
45
|
+
edge.from + 1, edge.to + 1, type, stereo)
|
|
46
|
+
end
|
|
47
|
+
|
|
48
|
+
lines << property_line("M CHG", charge_pairs(atoms))
|
|
49
|
+
lines << property_line("M ISO", isotope_pairs(atoms))
|
|
50
|
+
lines << "M END"
|
|
51
|
+
lines.compact.join("\n") << "\n"
|
|
52
|
+
end
|
|
53
|
+
|
|
54
|
+
private
|
|
55
|
+
|
|
56
|
+
# Authored coordinates win; otherwise positions from the
|
|
57
|
+
# deterministic 2D layout (same walk order as the graph).
|
|
58
|
+
def coordinates(atom, index, atoms)
|
|
59
|
+
return [atom.x2, atom.y2] if atom.x2 && atom.y2
|
|
60
|
+
|
|
61
|
+
@layout ||= AsciiChem::Layout.layout(@molecule)
|
|
62
|
+
placed = @layout.atoms[index]
|
|
63
|
+
placed ? [placed.x, placed.y] : [0.0, 0.0]
|
|
64
|
+
end
|
|
65
|
+
|
|
66
|
+
def charge_pairs(atoms)
|
|
67
|
+
atoms.each_with_index
|
|
68
|
+
.filter_map { |atom, i| [i + 1, charge_value(atom.charge)] if atom.charge }
|
|
69
|
+
end
|
|
70
|
+
|
|
71
|
+
def isotope_pairs(atoms)
|
|
72
|
+
atoms.each_with_index
|
|
73
|
+
.filter_map { |atom, i| [i + 1, atom.isotope.to_i] if atom.isotope }
|
|
74
|
+
end
|
|
75
|
+
|
|
76
|
+
# The model's number-then-sign charge ("2+") → signed integer.
|
|
77
|
+
def charge_value(charge)
|
|
78
|
+
count = charge[/\A\d+/]&.to_i || 1
|
|
79
|
+
charge.end_with?("-") ? -count : count
|
|
80
|
+
end
|
|
81
|
+
|
|
82
|
+
def property_line(prefix, pairs)
|
|
83
|
+
return nil if pairs.empty?
|
|
84
|
+
|
|
85
|
+
pairs.reduce(+"#{prefix}%3d" % pairs.length) do |line, (idx, value)|
|
|
86
|
+
line << format("%4d%4d", idx, value)
|
|
87
|
+
end
|
|
88
|
+
end
|
|
89
|
+
end
|
|
90
|
+
end
|
|
91
|
+
end
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module AsciiChem
|
|
4
|
+
# Molfile (CTfile V2000) ingestion and emission (TODO.v2 09;
|
|
5
|
+
# TODO.impl 57). Molfile is the highest-fidelity structure path:
|
|
6
|
+
# atom coordinates are preserved on the model (x2/y2/z2), charges
|
|
7
|
+
# via the CHG property block, isotopes via mass difference or the
|
|
8
|
+
# ISO block, bond stereo codes 1/6 map to wedge/hash bonds.
|
|
9
|
+
module Molfile
|
|
10
|
+
autoload :Parser, "asciichem/molfile/parser"
|
|
11
|
+
autoload :Writer, "asciichem/molfile/writer"
|
|
12
|
+
|
|
13
|
+
class << self
|
|
14
|
+
# Parses a V2000 molfile into a Model::Molecule.
|
|
15
|
+
def parse(text)
|
|
16
|
+
Parser.new(text).parse
|
|
17
|
+
end
|
|
18
|
+
|
|
19
|
+
# Emits a V2000 molfile. Uses authored x2/y2 coordinates;
|
|
20
|
+
# computes a 2D layout otherwise.
|
|
21
|
+
def write(molecule, name: nil)
|
|
22
|
+
Writer.new(molecule, name: name).write
|
|
23
|
+
end
|
|
24
|
+
end
|
|
25
|
+
end
|
|
26
|
+
end
|
|
@@ -0,0 +1,302 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module AsciiChem
|
|
4
|
+
module Smiles
|
|
5
|
+
# Recursive-descent parser for the supported SMILES subset.
|
|
6
|
+
# SMILES is a linear grammar, so a direct parser is clearer than
|
|
7
|
+
# a parse-tree + transform round trip; it builds the model
|
|
8
|
+
# through the shared Structure::Linearizer.
|
|
9
|
+
#
|
|
10
|
+
# v1 subset (each deferral raises an actionable ParseError):
|
|
11
|
+
# chirality `@`/`@@`, E/Z bond directions `/` `\`, wildcard `*`,
|
|
12
|
+
# reaction atom maps, queries.
|
|
13
|
+
class Parser
|
|
14
|
+
ORGANIC = %w[Cl Br B C N O P S F I].freeze
|
|
15
|
+
AROMATIC = %w[se as b c n o p s].freeze
|
|
16
|
+
|
|
17
|
+
def initialize(source)
|
|
18
|
+
@source = source
|
|
19
|
+
@pos = 0
|
|
20
|
+
end
|
|
21
|
+
|
|
22
|
+
# Returns a Model::Formula with one Molecule per dot component.
|
|
23
|
+
def parse
|
|
24
|
+
molecules = [component]
|
|
25
|
+
molecules << component while eat(".")
|
|
26
|
+
unless @pos == @source.length
|
|
27
|
+
raise ParseError,
|
|
28
|
+
"unexpected character #{peek.inspect} at position #{@pos} in #{@source.inspect}"
|
|
29
|
+
end
|
|
30
|
+
Model::Formula.new(nodes: molecules)
|
|
31
|
+
end
|
|
32
|
+
|
|
33
|
+
private
|
|
34
|
+
|
|
35
|
+
def component
|
|
36
|
+
@atoms = []
|
|
37
|
+
@adjacency = []
|
|
38
|
+
@open_rings = {}
|
|
39
|
+
@parent = nil
|
|
40
|
+
@pending_kind = nil
|
|
41
|
+
|
|
42
|
+
chain
|
|
43
|
+
|
|
44
|
+
unless @open_rings.empty?
|
|
45
|
+
raise ParseError,
|
|
46
|
+
"unclosed ring bond digit(s): #{@open_rings.keys.sort.join(', ')} in #{@source.inspect}"
|
|
47
|
+
end
|
|
48
|
+
|
|
49
|
+
edges = []
|
|
50
|
+
@adjacency.each_with_index do |neighbors, index|
|
|
51
|
+
neighbors.each { |to, kind| edges << Structure::Graph::Edge.new(from: index, to: to, kind: kind) if to > index }
|
|
52
|
+
end
|
|
53
|
+
Model::Molecule.new(nodes: Structure::Linearizer.new(atoms: @atoms, edges: edges).nodes)
|
|
54
|
+
end
|
|
55
|
+
|
|
56
|
+
# chain := atom (bond? atom | branch | ringbond)*
|
|
57
|
+
# Bare adjacency (no bond token) means a default bond.
|
|
58
|
+
def chain
|
|
59
|
+
atom_token
|
|
60
|
+
until eoc?
|
|
61
|
+
if peek == "("
|
|
62
|
+
branch
|
|
63
|
+
elsif bond_start?
|
|
64
|
+
kind = bond_token
|
|
65
|
+
if peek == "("
|
|
66
|
+
branch(kind)
|
|
67
|
+
elsif digit_start?
|
|
68
|
+
ringbond(kind)
|
|
69
|
+
else
|
|
70
|
+
@pending_kind = kind
|
|
71
|
+
atom_token
|
|
72
|
+
end
|
|
73
|
+
elsif digit_start?
|
|
74
|
+
ringbond(nil)
|
|
75
|
+
else
|
|
76
|
+
@pending_kind = nil
|
|
77
|
+
atom_token
|
|
78
|
+
end
|
|
79
|
+
end
|
|
80
|
+
end
|
|
81
|
+
|
|
82
|
+
def branch(kind = nil)
|
|
83
|
+
expect("(")
|
|
84
|
+
# The bond may lead the branch ("C-(O)") or open it ("C(=O)").
|
|
85
|
+
kind = bond_token if kind.nil? && bond_start?
|
|
86
|
+
saved_parent = @parent
|
|
87
|
+
saved_pending = @pending_kind
|
|
88
|
+
@pending_kind = kind
|
|
89
|
+
chain
|
|
90
|
+
expect(")")
|
|
91
|
+
@parent = saved_parent
|
|
92
|
+
@pending_kind = saved_pending
|
|
93
|
+
end
|
|
94
|
+
|
|
95
|
+
def ringbond(kind)
|
|
96
|
+
digit = ring_digit
|
|
97
|
+
if @open_rings.key?(digit)
|
|
98
|
+
opener = @open_rings.delete(digit)
|
|
99
|
+
# The model's ring-closure digits carry no bond kind, so an
|
|
100
|
+
# explicit kind that differs from the default (aromatic
|
|
101
|
+
# between aromatic atoms, otherwise single) cannot be
|
|
102
|
+
# represented — reject rather than silently downgrade.
|
|
103
|
+
if kind && kind != default_kind(opener, @parent)
|
|
104
|
+
raise ParseError,
|
|
105
|
+
"bonded ring closures (#{kind}) are not representable in the model's ring-closure form"
|
|
106
|
+
end
|
|
107
|
+
add_edge(opener, @parent, kind)
|
|
108
|
+
else
|
|
109
|
+
@open_rings[digit] = @parent
|
|
110
|
+
end
|
|
111
|
+
end
|
|
112
|
+
|
|
113
|
+
def atom_token
|
|
114
|
+
atom = read_atom
|
|
115
|
+
index = @atoms.length
|
|
116
|
+
@atoms << atom
|
|
117
|
+
@adjacency[index] = {}
|
|
118
|
+
add_edge(@parent, index, @pending_kind) if @parent
|
|
119
|
+
@parent = index
|
|
120
|
+
@pending_kind = nil
|
|
121
|
+
end
|
|
122
|
+
|
|
123
|
+
def add_edge(from, to, explicit)
|
|
124
|
+
kind = explicit || default_kind(from, to)
|
|
125
|
+
@adjacency[from][to] = kind
|
|
126
|
+
@adjacency[to][from] = kind
|
|
127
|
+
end
|
|
128
|
+
|
|
129
|
+
def default_kind(from, to)
|
|
130
|
+
@atoms[from].aromatic && @atoms[to].aromatic ? :aromatic : :single
|
|
131
|
+
end
|
|
132
|
+
|
|
133
|
+
# -- tokens ----------------------------------------------------------
|
|
134
|
+
|
|
135
|
+
def read_atom
|
|
136
|
+
return bracket_atom if peek == "["
|
|
137
|
+
|
|
138
|
+
two = @source[@pos, 2].to_s
|
|
139
|
+
if (symbol = ORGANIC.find { |s| two.start_with?(s) })
|
|
140
|
+
@pos += symbol.length
|
|
141
|
+
return Model::Atom.new(element: symbol)
|
|
142
|
+
end
|
|
143
|
+
if (symbol = AROMATIC.find { |s| two.start_with?(s) } || AROMATIC.find { |s| peek == s })
|
|
144
|
+
@pos += symbol.length
|
|
145
|
+
return Model::Atom.new(element: symbol.capitalize, aromatic: true)
|
|
146
|
+
end
|
|
147
|
+
|
|
148
|
+
raise ParseError, "unexpected character #{peek.inspect} at position #{@pos} in #{@source.inspect}"
|
|
149
|
+
end
|
|
150
|
+
|
|
151
|
+
def bracket_atom
|
|
152
|
+
expect("[")
|
|
153
|
+
isotope = digits
|
|
154
|
+
symbol = bracket_symbol
|
|
155
|
+
reject_chirality
|
|
156
|
+
hydrogens = hcount
|
|
157
|
+
charge = bracket_charge
|
|
158
|
+
skip_class
|
|
159
|
+
expect("]")
|
|
160
|
+
Model::Atom.new(
|
|
161
|
+
element: symbol.capitalize,
|
|
162
|
+
isotope: isotope,
|
|
163
|
+
charge: charge,
|
|
164
|
+
hydrogens: hydrogens,
|
|
165
|
+
aromatic: symbol.match?(/\A[a-z]/) || nil
|
|
166
|
+
)
|
|
167
|
+
end
|
|
168
|
+
|
|
169
|
+
AROMATIC_TWO_CHAR = %w[se as].freeze
|
|
170
|
+
|
|
171
|
+
def bracket_symbol
|
|
172
|
+
two = @source[@pos, 2].to_s
|
|
173
|
+
if AROMATIC_TWO_CHAR.include?(two)
|
|
174
|
+
@pos += 2
|
|
175
|
+
return two
|
|
176
|
+
end
|
|
177
|
+
if @source[@pos] =~ /[bcnops]/
|
|
178
|
+
sym = @source[@pos]
|
|
179
|
+
@pos += 1
|
|
180
|
+
return sym
|
|
181
|
+
end
|
|
182
|
+
if @source[@pos] =~ /[A-Z]/
|
|
183
|
+
sym = @source[@pos]
|
|
184
|
+
sym += @source[@pos + 1] if @source[@pos + 1] =~ /[a-z]/
|
|
185
|
+
@pos += sym.length
|
|
186
|
+
return sym
|
|
187
|
+
end
|
|
188
|
+
raise ParseError, "expected an element symbol at position #{@pos} in #{@source.inspect}"
|
|
189
|
+
end
|
|
190
|
+
|
|
191
|
+
def reject_chirality
|
|
192
|
+
return unless peek == "@"
|
|
193
|
+
|
|
194
|
+
token = @source[@pos, 2] == "@@" ? "'@@'" : "'@'"
|
|
195
|
+
raise ParseError, "chirality #{token} is not supported in the v1 subset"
|
|
196
|
+
end
|
|
197
|
+
|
|
198
|
+
def hcount
|
|
199
|
+
return nil unless peek == "H"
|
|
200
|
+
|
|
201
|
+
@pos += 1
|
|
202
|
+
count = digits
|
|
203
|
+
count ? count.to_i : 1
|
|
204
|
+
end
|
|
205
|
+
|
|
206
|
+
# "+" | "++" | "+n" | "-" | "--" | "-n" → number-then-sign.
|
|
207
|
+
def bracket_charge
|
|
208
|
+
sign = peek
|
|
209
|
+
return nil unless sign == "+" || sign == "-"
|
|
210
|
+
|
|
211
|
+
@pos += 1
|
|
212
|
+
second = @source[@pos].to_s
|
|
213
|
+
count =
|
|
214
|
+
if second == sign
|
|
215
|
+
@pos += 1
|
|
216
|
+
2
|
|
217
|
+
elsif second =~ /[0-9]/
|
|
218
|
+
digits.to_i
|
|
219
|
+
end
|
|
220
|
+
count&.positive? ? "#{count}#{sign}" : sign
|
|
221
|
+
end
|
|
222
|
+
|
|
223
|
+
def skip_class
|
|
224
|
+
return unless peek == ":"
|
|
225
|
+
|
|
226
|
+
@pos += 1
|
|
227
|
+
digits
|
|
228
|
+
end
|
|
229
|
+
|
|
230
|
+
def bond_token
|
|
231
|
+
ch = peek
|
|
232
|
+
kind = { "-" => :single, "=" => :double, "#" => :triple, "$" => :quadruple,
|
|
233
|
+
":" => :aromatic }[ch]
|
|
234
|
+
unless kind
|
|
235
|
+
if ch == "/" || ch == "\\"
|
|
236
|
+
raise ParseError,
|
|
237
|
+
"bond direction #{ch.inspect} (E/Z stereo) is not supported in the v1 subset"
|
|
238
|
+
end
|
|
239
|
+
raise ParseError, "expected a bond or atom at position #{@pos} in #{@source.inspect}"
|
|
240
|
+
end
|
|
241
|
+
@pos += 1
|
|
242
|
+
kind
|
|
243
|
+
end
|
|
244
|
+
|
|
245
|
+
def ring_digit
|
|
246
|
+
if peek == "%"
|
|
247
|
+
@pos += 1
|
|
248
|
+
digit = @source[@pos, 2].to_s
|
|
249
|
+
raise ParseError, "malformed %nn ring closure" unless digit.match?(/\A\d\d\z/)
|
|
250
|
+
|
|
251
|
+
@pos += 2
|
|
252
|
+
digit
|
|
253
|
+
else
|
|
254
|
+
d = peek
|
|
255
|
+
raise ParseError, "expected ring digit at position #{@pos}" unless d =~ /[0-9]/
|
|
256
|
+
|
|
257
|
+
@pos += 1
|
|
258
|
+
d
|
|
259
|
+
end
|
|
260
|
+
end
|
|
261
|
+
|
|
262
|
+
# -- character helpers -------------------------------------------------
|
|
263
|
+
|
|
264
|
+
def peek
|
|
265
|
+
@source[@pos]
|
|
266
|
+
end
|
|
267
|
+
|
|
268
|
+
def digits
|
|
269
|
+
start = @pos
|
|
270
|
+
@pos += 1 while @source[@pos] =~ /[0-9]/
|
|
271
|
+
@pos == start ? nil : @source[start...@pos]
|
|
272
|
+
end
|
|
273
|
+
|
|
274
|
+
def bond_start?
|
|
275
|
+
%w[- = # $ : / \\].include?(peek)
|
|
276
|
+
end
|
|
277
|
+
|
|
278
|
+
def digit_start?
|
|
279
|
+
peek =~ /[0-9%]/
|
|
280
|
+
end
|
|
281
|
+
|
|
282
|
+
def eoc?
|
|
283
|
+
peek.nil? || %w[. )].include?(peek)
|
|
284
|
+
end
|
|
285
|
+
|
|
286
|
+
def eat(char)
|
|
287
|
+
return false unless peek == char
|
|
288
|
+
|
|
289
|
+
@pos += 1
|
|
290
|
+
true
|
|
291
|
+
end
|
|
292
|
+
|
|
293
|
+
def expect(char)
|
|
294
|
+
unless peek == char
|
|
295
|
+
raise ParseError, "expected #{char.inspect} at position #{@pos} in #{@source.inspect}"
|
|
296
|
+
end
|
|
297
|
+
|
|
298
|
+
@pos += 1
|
|
299
|
+
end
|
|
300
|
+
end
|
|
301
|
+
end
|
|
302
|
+
end
|
|
@@ -0,0 +1,210 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module AsciiChem
|
|
4
|
+
module Smiles
|
|
5
|
+
# Model molecule → deterministic SMILES.
|
|
6
|
+
#
|
|
7
|
+
# Deterministic rules (not a canonical-rank algorithm):
|
|
8
|
+
# - DFS from the first atom.
|
|
9
|
+
# - At each atom: ring-closure digits first, then branches
|
|
10
|
+
# (sorted), then the continuation chain.
|
|
11
|
+
# - The continuation child is chosen by: single/aromatic bond
|
|
12
|
+
# first (multiple bonds become branches, the way chemists write
|
|
13
|
+
# SMILES), then the largest unvisited subtree, then the lowest
|
|
14
|
+
# index.
|
|
15
|
+
# - Ring digits are assigned in encounter order (1..9).
|
|
16
|
+
#
|
|
17
|
+
# The stability property `write(parse(write(x))) == write(x)` and
|
|
18
|
+
# structural round-tripping are spec'd. Wedge/hash/dative/wavy
|
|
19
|
+
# bonds and formula-only molecules raise.
|
|
20
|
+
class Writer
|
|
21
|
+
LOWERCASE_AROMATIC = %w[C N O S P B Se As].freeze
|
|
22
|
+
ORGANIC = %w[B C N O P S F Cl Br I].freeze
|
|
23
|
+
BOND_TOKENS = {
|
|
24
|
+
single: "-", double: "=", triple: "#", quadruple: "$", aromatic: ":"
|
|
25
|
+
}.freeze
|
|
26
|
+
CONTINUATION_BONDS = %i[single aromatic].freeze
|
|
27
|
+
|
|
28
|
+
def initialize(molecule)
|
|
29
|
+
@molecule = molecule
|
|
30
|
+
end
|
|
31
|
+
|
|
32
|
+
def write
|
|
33
|
+
atoms, edges = Structure::Graph.build(@molecule)
|
|
34
|
+
# A single-atom component is valid SMILES (water "O", ions);
|
|
35
|
+
# multiple atoms with no bonds is a formula, not a structure.
|
|
36
|
+
if edges.empty? && atoms.length > 1
|
|
37
|
+
raise ParseError, "molecule has no bonds — a formula is not a structure"
|
|
38
|
+
end
|
|
39
|
+
|
|
40
|
+
@atoms = atoms
|
|
41
|
+
build_adjacency(edges)
|
|
42
|
+
@visited = {}
|
|
43
|
+
@digits = Array.new(atoms.length) { [] }
|
|
44
|
+
@digit_by_edge = {}
|
|
45
|
+
@next_digit = 1
|
|
46
|
+
# Emission builds ordered units (atoms, literals); ring digits
|
|
47
|
+
# attach to per-atom lists so both endpoints carry them
|
|
48
|
+
# without string surgery.
|
|
49
|
+
@units = []
|
|
50
|
+
@tree_children = Array.new(atoms.length) { [] }
|
|
51
|
+
@closures = Array.new(atoms.length) { [] }
|
|
52
|
+
@tree_size = Array.new(atoms.length, 1)
|
|
53
|
+
|
|
54
|
+
build_tree(0, nil)
|
|
55
|
+
compute_tree_size(0)
|
|
56
|
+
emit(0, nil, nil)
|
|
57
|
+
|
|
58
|
+
@units.map do |unit|
|
|
59
|
+
if unit.first == :atom
|
|
60
|
+
index = unit.last
|
|
61
|
+
atom_token(index) + @digits[index].join
|
|
62
|
+
else
|
|
63
|
+
unit.last
|
|
64
|
+
end
|
|
65
|
+
end.join
|
|
66
|
+
end
|
|
67
|
+
|
|
68
|
+
private
|
|
69
|
+
|
|
70
|
+
def build_adjacency(edges)
|
|
71
|
+
@adjacency = Array.new(@atoms.length) { {} }
|
|
72
|
+
edges.each do |edge|
|
|
73
|
+
@adjacency[edge.from][edge.to] = edge.kind
|
|
74
|
+
@adjacency[edge.to][edge.from] = edge.kind
|
|
75
|
+
end
|
|
76
|
+
end
|
|
77
|
+
|
|
78
|
+
# Post-order subtree sizes within the DFS tree.
|
|
79
|
+
def compute_tree_size(index)
|
|
80
|
+
@tree_children[index].each { |child| compute_tree_size(child) }
|
|
81
|
+
@tree_size[index] = 1 + @tree_children[index].sum { |child| @tree_size[child] }
|
|
82
|
+
end
|
|
83
|
+
|
|
84
|
+
# First pass: a deterministic DFS tree. Children are visited in
|
|
85
|
+
# index order; edges to already-visited atoms become closure
|
|
86
|
+
# digits. Recording the tree first means a branch can never
|
|
87
|
+
# walk around a ring and swallow the continuation atom.
|
|
88
|
+
def build_tree(index, parent)
|
|
89
|
+
@visited[index] = true
|
|
90
|
+
@adjacency[index].keys.sort.each do |nb|
|
|
91
|
+
if nb != parent && @visited[nb]
|
|
92
|
+
@closures[index] << nb
|
|
93
|
+
else
|
|
94
|
+
next if @visited[nb]
|
|
95
|
+
|
|
96
|
+
@tree_children[index] << nb
|
|
97
|
+
build_tree(nb, index)
|
|
98
|
+
end
|
|
99
|
+
end
|
|
100
|
+
end
|
|
101
|
+
|
|
102
|
+
def emit(index, parent, incoming_kind)
|
|
103
|
+
@units << [:literal, bond_token(incoming_kind, parent, index)] if parent
|
|
104
|
+
@units << [:atom, index]
|
|
105
|
+
@closures[index].each { |nb| ring_digit_for(index, nb) }
|
|
106
|
+
|
|
107
|
+
children = @tree_children[index]
|
|
108
|
+
return if children.empty?
|
|
109
|
+
|
|
110
|
+
ordered = children.sort_by { |child| continuation_rank(index, child) }
|
|
111
|
+
continuation, *branches = ordered
|
|
112
|
+
|
|
113
|
+
branches.each do |child|
|
|
114
|
+
@units << [:literal, "("]
|
|
115
|
+
emit(child, index, @adjacency[index][child])
|
|
116
|
+
@units << [:literal, ")"]
|
|
117
|
+
end
|
|
118
|
+
emit(continuation, index, @adjacency[index][continuation])
|
|
119
|
+
end
|
|
120
|
+
|
|
121
|
+
# Lower sorts first: single/aromatic bonds continue the chain,
|
|
122
|
+
# then larger subtrees, then the lexicographically smallest atom
|
|
123
|
+
# token (order-independent, so the canonical form does not
|
|
124
|
+
# depend on how the input happened to number the atoms).
|
|
125
|
+
def continuation_rank(index, child)
|
|
126
|
+
kind = @adjacency[index][child]
|
|
127
|
+
priority = CONTINUATION_BONDS.include?(kind) ? 0 : 1
|
|
128
|
+
[priority, -@tree_size[child], atom_token(child)]
|
|
129
|
+
end
|
|
130
|
+
|
|
131
|
+
def ring_digit_for(a, b)
|
|
132
|
+
key = [a, b].minmax
|
|
133
|
+
digit = @digit_by_edge[key]
|
|
134
|
+
return digit if digit
|
|
135
|
+
|
|
136
|
+
raise ParseError, "too many ring closures for SMILES output" if @next_digit > 9
|
|
137
|
+
|
|
138
|
+
digit = @next_digit.to_s
|
|
139
|
+
@next_digit += 1
|
|
140
|
+
@digit_by_edge[key] = digit
|
|
141
|
+
# Both atom tokens carry the digit; each endpoint's list
|
|
142
|
+
# receives it in encounter order.
|
|
143
|
+
@digits[key.first] << digit
|
|
144
|
+
@digits[key.last] << digit
|
|
145
|
+
digit
|
|
146
|
+
end
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def bond_token(kind, from, to)
|
|
150
|
+
token = BOND_TOKENS[kind]
|
|
151
|
+
raise ParseError, "#{kind} bonds have no SMILES form (v1 subset)" unless token
|
|
152
|
+
|
|
153
|
+
return token if kind != :single && kind != :aromatic
|
|
154
|
+
|
|
155
|
+
# Explicit only when the short form would change the meaning:
|
|
156
|
+
# a single bond between two aromatic atoms must be written;
|
|
157
|
+
# an aromatic bond is implicit between two aromatic atoms.
|
|
158
|
+
if kind == :single
|
|
159
|
+
aromatic?(from) && aromatic?(to) ? token : ""
|
|
160
|
+
else
|
|
161
|
+
aromatic?(from) && aromatic?(to) ? "" : token
|
|
162
|
+
end
|
|
163
|
+
end
|
|
164
|
+
|
|
165
|
+
def atom_token(index)
|
|
166
|
+
@atom_tokens ||= {}
|
|
167
|
+
@atom_tokens[index] ||= begin
|
|
168
|
+
atom = @atoms[index]
|
|
169
|
+
aromatic = atom.aromatic == true
|
|
170
|
+
lowercase = aromatic && LOWERCASE_AROMATIC.include?(atom.element)
|
|
171
|
+
needs_bracket =
|
|
172
|
+
atom.charge || atom.isotope || atom.hydrogens ||
|
|
173
|
+
(!ORGANIC.include?(atom.element) && !lowercase) ||
|
|
174
|
+
(aromatic && !lowercase)
|
|
175
|
+
unless needs_bracket
|
|
176
|
+
bare_symbol(atom, lowercase)
|
|
177
|
+
else
|
|
178
|
+
symbol = lowercase ? atom.element.downcase : atom.element
|
|
179
|
+
token = +"["
|
|
180
|
+
token << atom.isotope if atom.isotope
|
|
181
|
+
token << symbol
|
|
182
|
+
if atom.hydrogens
|
|
183
|
+
token << "H"
|
|
184
|
+
token << atom.hydrogens.to_s if atom.hydrogens > 1
|
|
185
|
+
end
|
|
186
|
+
token << charge_suffix(atom.charge) if atom.charge
|
|
187
|
+
token << "]"
|
|
188
|
+
end
|
|
189
|
+
end
|
|
190
|
+
end
|
|
191
|
+
|
|
192
|
+
def bare_symbol(atom, lowercase)
|
|
193
|
+
lowercase ? atom.element.downcase : atom.element
|
|
194
|
+
end
|
|
195
|
+
|
|
196
|
+
# The model's number-then-sign charge ("2+") → SMILES ("+2").
|
|
197
|
+
def charge_suffix(charge)
|
|
198
|
+
count = charge[/\A\d+/]
|
|
199
|
+
sign = charge[-1]
|
|
200
|
+
return sign unless count && count.to_i > 1
|
|
201
|
+
|
|
202
|
+
"#{sign}#{count}"
|
|
203
|
+
end
|
|
204
|
+
|
|
205
|
+
def aromatic?(index)
|
|
206
|
+
@atoms[index].aromatic == true
|
|
207
|
+
end
|
|
208
|
+
end
|
|
209
|
+
end
|
|
210
|
+
end
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "parslet"
|
|
4
|
+
|
|
5
|
+
module AsciiChem
|
|
6
|
+
# SMILES ingestion and emission for the semantic model
|
|
7
|
+
# (TODO.v2 09; TODO.impl 57). SMILES is an input syntax for the one
|
|
8
|
+
# semantic model — parsed molecules are ordinary `Model::Molecule`s
|
|
9
|
+
# with explicit bonds and ring closures, renderable by every
|
|
10
|
+
# formatter.
|
|
11
|
+
#
|
|
12
|
+
# v1 subset (each deferral has a rejecting spec with an actionable
|
|
13
|
+
# message): chirality `@`/`@@` and E/Z bond directions `/` `\` are
|
|
14
|
+
# not parsed; reaction atom maps, queries, and S-groups are out of
|
|
15
|
+
# scope.
|
|
16
|
+
module Smiles
|
|
17
|
+
autoload :Parser, "asciichem/smiles/parser"
|
|
18
|
+
autoload :Writer, "asciichem/smiles/writer"
|
|
19
|
+
|
|
20
|
+
class << self
|
|
21
|
+
# Parses a SMILES string into a Model::Formula with one
|
|
22
|
+
# Molecule per dot-disconnected component.
|
|
23
|
+
def parse(smiles)
|
|
24
|
+
Parser.new(smiles).parse
|
|
25
|
+
end
|
|
26
|
+
|
|
27
|
+
# Emits deterministic SMILES (DFS from the first atom,
|
|
28
|
+
# neighbours in creation order, ring digits assigned in
|
|
29
|
+
# encounter order). Not a canonical-rank algorithm; the
|
|
30
|
+
# stability property `write(parse(write(x))) == write(x)`
|
|
31
|
+
# holds and is spec'd.
|
|
32
|
+
def write(node)
|
|
33
|
+
case node
|
|
34
|
+
when AsciiChem::Model::Formula
|
|
35
|
+
node.nodes.map { |m| Writer.new(m).write }.join(".")
|
|
36
|
+
when AsciiChem::Model::Molecule
|
|
37
|
+
Writer.new(node).write
|
|
38
|
+
else
|
|
39
|
+
raise AsciiChem::ParseError,
|
|
40
|
+
"#{node.class} has no SMILES form (molecules and formulas of molecules only)"
|
|
41
|
+
end
|
|
42
|
+
end
|
|
43
|
+
end
|
|
44
|
+
end
|
|
45
|
+
end
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module AsciiChem
|
|
4
|
+
module Structure
|
|
5
|
+
# Model molecule → neutral graph. Bond tokens between consecutive
|
|
6
|
+
# atoms become edges (pending-bond semantics, same as the Layout
|
|
7
|
+
# walker); ring-closure digits become edges via AsciiChem::RingBonds.
|
|
8
|
+
module Graph
|
|
9
|
+
Edge = Struct.new(:from, :to, :kind, keyword_init: true)
|
|
10
|
+
|
|
11
|
+
class << self
|
|
12
|
+
# Returns [atoms, edges] where atoms is the atom array in
|
|
13
|
+
# walk order and edges is an array of Edge with array indexes.
|
|
14
|
+
def build(molecule)
|
|
15
|
+
atoms = []
|
|
16
|
+
edges = []
|
|
17
|
+
index_by_object_id = {}
|
|
18
|
+
pending = nil
|
|
19
|
+
last = nil
|
|
20
|
+
|
|
21
|
+
walk = lambda do |nodes|
|
|
22
|
+
nodes.each do |node|
|
|
23
|
+
case node
|
|
24
|
+
when AsciiChem::Model::Atom
|
|
25
|
+
index = atoms.length
|
|
26
|
+
# Atoms override == with value equality (benzene's
|
|
27
|
+
# carbons are all equal), so identity is by object_id —
|
|
28
|
+
# same as the Layout walker.
|
|
29
|
+
index_by_object_id[node.object_id] = index
|
|
30
|
+
atoms << node
|
|
31
|
+
if pending && last
|
|
32
|
+
edges << Edge.new(from: last, to: index, kind: pending.kind)
|
|
33
|
+
end
|
|
34
|
+
last = index
|
|
35
|
+
pending = nil
|
|
36
|
+
when AsciiChem::Model::Bond
|
|
37
|
+
pending = node
|
|
38
|
+
when AsciiChem::Model::Group, AsciiChem::Model::Molecule
|
|
39
|
+
walk.call(node.nodes)
|
|
40
|
+
end
|
|
41
|
+
end
|
|
42
|
+
end
|
|
43
|
+
walk.call(molecule.nodes)
|
|
44
|
+
|
|
45
|
+
AsciiChem::RingBonds.each_in(molecule) do |ring|
|
|
46
|
+
from = index_by_object_id[ring.from_atom.object_id]
|
|
47
|
+
to = index_by_object_id[ring.to_atom.object_id]
|
|
48
|
+
next unless from && to
|
|
49
|
+
|
|
50
|
+
# Ring-closure digits carry no kind; the default bond rule
|
|
51
|
+
# (aromatic between aromatic atoms, otherwise single)
|
|
52
|
+
# applies — the same rule the SMILES parser uses.
|
|
53
|
+
kind = ring.from_atom.aromatic && ring.to_atom.aromatic ? :aromatic : :single
|
|
54
|
+
edges << Edge.new(from: from, to: to, kind: kind)
|
|
55
|
+
end
|
|
56
|
+
|
|
57
|
+
[atoms, edges]
|
|
58
|
+
end
|
|
59
|
+
end
|
|
60
|
+
end
|
|
61
|
+
end
|
|
62
|
+
end
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module AsciiChem
|
|
4
|
+
module Structure
|
|
5
|
+
# Adjacency → linear model nodes. Atoms are emitted in the
|
|
6
|
+
# caller's creation order (a DFS order for parsers that walk
|
|
7
|
+
# depth-first). An edge between consecutive atoms becomes a
|
|
8
|
+
# `Bond` token; every other edge becomes a pair of matching ring
|
|
9
|
+
# closure digits — exactly the mechanism AsciiChem's own ring
|
|
10
|
+
# syntax uses, so downstream walkers (`RingBonds`, Layout,
|
|
11
|
+
# ModelAdapter) see the graph with no new machinery.
|
|
12
|
+
#
|
|
13
|
+
# Ring digits are allocated 1..9 and reused only once the
|
|
14
|
+
# previous interval (first endpoint .. second endpoint, in atom
|
|
15
|
+
# order) has closed: `RingBonds` pairs digits by occurrence order
|
|
16
|
+
# across the molecule, so overlapping intervals must never share
|
|
17
|
+
# a digit.
|
|
18
|
+
class Linearizer
|
|
19
|
+
MAX_DIGIT = 9
|
|
20
|
+
|
|
21
|
+
def initialize(atoms:, edges:)
|
|
22
|
+
@atoms = atoms
|
|
23
|
+
@edges = edges
|
|
24
|
+
@close_at = Array.new(MAX_DIGIT, -1)
|
|
25
|
+
@digits_for = Array.new(@atoms.length) { "" }
|
|
26
|
+
@token_before = {}
|
|
27
|
+
end
|
|
28
|
+
|
|
29
|
+
# Returns the node array (Atoms and Bonds interleaved) and
|
|
30
|
+
# mutates the atoms' `ring_closures` with the allocated digits.
|
|
31
|
+
def nodes
|
|
32
|
+
@edges.each do |edge|
|
|
33
|
+
first, last = [edge.from, edge.to].minmax
|
|
34
|
+
if last - first == 1
|
|
35
|
+
@token_before[last] = edge.kind
|
|
36
|
+
else
|
|
37
|
+
digit = allocate_digit(first, last)
|
|
38
|
+
@digits_for[first] += digit
|
|
39
|
+
@digits_for[last] += digit
|
|
40
|
+
end
|
|
41
|
+
end
|
|
42
|
+
|
|
43
|
+
result = []
|
|
44
|
+
@atoms.each_with_index do |atom, index|
|
|
45
|
+
kind = @token_before[index]
|
|
46
|
+
result << AsciiChem::Model::Bond.new(kind: kind || :single) if index.positive? && kind
|
|
47
|
+
atom.ring_closures = @digits_for[index] unless @digits_for[index].empty?
|
|
48
|
+
result << atom
|
|
49
|
+
end
|
|
50
|
+
result
|
|
51
|
+
end
|
|
52
|
+
|
|
53
|
+
private
|
|
54
|
+
|
|
55
|
+
def allocate_digit(first, last)
|
|
56
|
+
digit_index = (0...MAX_DIGIT).find { |d| @close_at[d] <= first }
|
|
57
|
+
unless digit_index
|
|
58
|
+
raise AsciiChem::ParseError,
|
|
59
|
+
"more than #{MAX_DIGIT} overlapping non-adjacent bonds — " \
|
|
60
|
+
"beyond the model's ring-closure digit capacity"
|
|
61
|
+
end
|
|
62
|
+
@close_at[digit_index] = last
|
|
63
|
+
(digit_index + 1).to_s
|
|
64
|
+
end
|
|
65
|
+
end
|
|
66
|
+
end
|
|
67
|
+
end
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module AsciiChem
|
|
4
|
+
# Shared structure plumbing for the interchange formats (SMILES,
|
|
5
|
+
# molfile). Two MECE concerns:
|
|
6
|
+
#
|
|
7
|
+
# - `Structure::Graph.build(molecule)` — model molecule → neutral
|
|
8
|
+
# atom list + bond list (adjacency), using the same pending-bond
|
|
9
|
+
# walk semantics as the Layout walker plus `RingBonds` closure
|
|
10
|
+
# edges.
|
|
11
|
+
# - `Structure::Linearizer` — adjacency → linear model nodes (atoms
|
|
12
|
+
# in creation order, `Bond` tokens for consecutive edges, ring
|
|
13
|
+
# closure digits for non-consecutive edges). This is what lets an
|
|
14
|
+
# arbitrary graph from a database live inside the existing model
|
|
15
|
+
# and render through every existing formatter.
|
|
16
|
+
module Structure
|
|
17
|
+
autoload :Graph, "asciichem/structure/graph"
|
|
18
|
+
autoload :Linearizer, "asciichem/structure/linearizer"
|
|
19
|
+
end
|
|
20
|
+
end
|
data/lib/asciichem/version.rb
CHANGED
data/lib/asciichem/wire/core.rb
CHANGED
|
@@ -16,6 +16,8 @@ module AsciiChem
|
|
|
16
16
|
attribute :lone_pairs, :integer
|
|
17
17
|
attribute :radical_electrons, :integer
|
|
18
18
|
attribute :ring_closures, :string
|
|
19
|
+
attribute :aromatic, :boolean
|
|
20
|
+
attribute :hydrogens, :integer
|
|
19
21
|
json do
|
|
20
22
|
map "element", to: :element
|
|
21
23
|
map "isotope", to: :isotope
|
|
@@ -25,6 +27,8 @@ module AsciiChem
|
|
|
25
27
|
map "lonePairs", to: :lone_pairs
|
|
26
28
|
map "radicalElectrons", to: :radical_electrons
|
|
27
29
|
map "ringClosures", to: :ring_closures
|
|
30
|
+
map "aromatic", to: :aromatic
|
|
31
|
+
map "hydrogens", to: :hydrogens
|
|
28
32
|
end
|
|
29
33
|
end
|
|
30
34
|
|
|
@@ -105,7 +105,9 @@ module AsciiChem
|
|
|
105
105
|
oxidation_state: str_or_nil(atom.oxidation_state),
|
|
106
106
|
lone_pairs: int_or_nil(atom.lone_pairs),
|
|
107
107
|
radical_electrons: int_or_nil(atom.radical_electrons),
|
|
108
|
-
ring_closures: str_or_nil(atom.ring_closures)
|
|
108
|
+
ring_closures: str_or_nil(atom.ring_closures),
|
|
109
|
+
aromatic: bool_or_nil(atom.aromatic),
|
|
110
|
+
hydrogens: int_or_nil(atom.hydrogens)
|
|
109
111
|
)
|
|
110
112
|
end
|
|
111
113
|
|
|
@@ -121,6 +123,10 @@ module AsciiChem
|
|
|
121
123
|
value.is_a?(String) ? value : nil
|
|
122
124
|
end
|
|
123
125
|
|
|
126
|
+
def bool_or_nil(value)
|
|
127
|
+
value.nil? ? nil : !!value
|
|
128
|
+
end
|
|
129
|
+
|
|
124
130
|
def int_or_nil(value)
|
|
125
131
|
value.is_a?(Integer) ? value : nil
|
|
126
132
|
end
|
|
@@ -280,7 +286,8 @@ module AsciiChem
|
|
|
280
286
|
element: wire.element, isotope: wire.isotope, subscript: wire.subscript,
|
|
281
287
|
charge: wire.charge, oxidation_state: wire.oxidation_state,
|
|
282
288
|
lone_pairs: wire.lone_pairs, radical_electrons: wire.radical_electrons,
|
|
283
|
-
ring_closures: wire.ring_closures
|
|
289
|
+
ring_closures: wire.ring_closures,
|
|
290
|
+
aromatic: wire.aromatic, hydrogens: wire.hydrogens
|
|
284
291
|
)
|
|
285
292
|
end
|
|
286
293
|
|
data/lib/asciichem.rb
CHANGED
|
@@ -22,9 +22,12 @@ module AsciiChem
|
|
|
22
22
|
autoload :Linter, "asciichem/linter"
|
|
23
23
|
autoload :Model, "asciichem/model"
|
|
24
24
|
autoload :ModelAdapter, "asciichem/model_adapter"
|
|
25
|
+
autoload :Molfile, "asciichem/molfile"
|
|
25
26
|
autoload :Parser, "asciichem/parser"
|
|
26
27
|
autoload :PeriodicTable, "asciichem/periodic_table"
|
|
27
28
|
autoload :RingBonds, "asciichem/ring_bonds"
|
|
29
|
+
autoload :Smiles, "asciichem/smiles"
|
|
30
|
+
autoload :Structure, "asciichem/structure"
|
|
28
31
|
autoload :Transform, "asciichem/transform"
|
|
29
32
|
autoload :VERSION, "asciichem/version"
|
|
30
33
|
autoload :Wire, "asciichem/wire"
|
|
@@ -41,4 +44,17 @@ module AsciiChem
|
|
|
41
44
|
def self.from_model_json(json)
|
|
42
45
|
WireAdapter.from_model_json(json)
|
|
43
46
|
end
|
|
47
|
+
|
|
48
|
+
# Ingests a SMILES string into the semantic model (TODO.v2 09):
|
|
49
|
+
# one Model::Molecule per dot-disconnected component, explicit
|
|
50
|
+
# bonds and ring closures, renderable by every formatter.
|
|
51
|
+
def self.parse_smiles(smiles)
|
|
52
|
+
Smiles.parse(smiles)
|
|
53
|
+
end
|
|
54
|
+
|
|
55
|
+
# Ingests a molfile (CTfile V2000) into a Model::Molecule,
|
|
56
|
+
# preserving atom coordinates.
|
|
57
|
+
def self.parse_molfile(text)
|
|
58
|
+
Molfile.parse(text)
|
|
59
|
+
end
|
|
44
60
|
end
|
metadata
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
--- !ruby/object:Gem::Specification
|
|
2
2
|
name: asciichem
|
|
3
3
|
version: !ruby/object:Gem::Version
|
|
4
|
-
version: 0.
|
|
4
|
+
version: 0.21.0
|
|
5
5
|
platform: ruby
|
|
6
6
|
authors:
|
|
7
7
|
- Ribose Inc.
|
|
@@ -223,9 +223,18 @@ files:
|
|
|
223
223
|
- lib/asciichem/model_adapter.rb
|
|
224
224
|
- lib/asciichem/model_adapter/from_canonical.rb
|
|
225
225
|
- lib/asciichem/model_adapter/to_canonical.rb
|
|
226
|
+
- lib/asciichem/molfile.rb
|
|
227
|
+
- lib/asciichem/molfile/parser.rb
|
|
228
|
+
- lib/asciichem/molfile/writer.rb
|
|
226
229
|
- lib/asciichem/parser.rb
|
|
227
230
|
- lib/asciichem/periodic_table.rb
|
|
228
231
|
- lib/asciichem/ring_bonds.rb
|
|
232
|
+
- lib/asciichem/smiles.rb
|
|
233
|
+
- lib/asciichem/smiles/parser.rb
|
|
234
|
+
- lib/asciichem/smiles/writer.rb
|
|
235
|
+
- lib/asciichem/structure.rb
|
|
236
|
+
- lib/asciichem/structure/graph.rb
|
|
237
|
+
- lib/asciichem/structure/linearizer.rb
|
|
229
238
|
- lib/asciichem/transform.rb
|
|
230
239
|
- lib/asciichem/version.rb
|
|
231
240
|
- lib/asciichem/wire.rb
|