asciichem 0.20.0 → 0.22.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +39 -1
- data/Gemfile +1 -0
- data/asciichem.gemspec +1 -1
- data/lib/asciichem/cli.rb +34 -5
- data/lib/asciichem/formatter/structural_svg.rb +8 -0
- data/lib/asciichem/model/atom.rb +8 -3
- data/lib/asciichem/model/bond.rb +2 -1
- data/lib/asciichem/model/node.rb +12 -0
- data/lib/asciichem/molfile/parser.rb +169 -0
- data/lib/asciichem/molfile/writer.rb +91 -0
- data/lib/asciichem/molfile.rb +26 -0
- data/lib/asciichem/smiles/parser.rb +302 -0
- data/lib/asciichem/smiles/writer.rb +210 -0
- data/lib/asciichem/smiles.rb +45 -0
- data/lib/asciichem/structure/graph.rb +62 -0
- data/lib/asciichem/structure/linearizer.rb +67 -0
- data/lib/asciichem/structure.rb +20 -0
- data/lib/asciichem/version.rb +1 -1
- data/lib/asciichem/wire/core.rb +4 -0
- data/lib/asciichem/wire_adapter.rb +9 -2
- data/lib/asciichem.rb +16 -0
- data/scripts/update-model-schemas.sh +26 -0
- metadata +14 -4
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: d1aef5e42c0840b8bff4d291220e84f76f9536a9dba1cca3445deacb696388b1
|
|
4
|
+
data.tar.gz: 92e28e623dfb7494080a9ee80d3545fa5a593d83d545775d890b7a9a5594c0bd
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: bab5532c9007d1a5bd2079b792a66458d411f48af5ac59f04ad9e4a9150c7b53844639e67a46494dac7e6ae05f3451904a8fbe927b711803184749f00ce40e7e
|
|
7
|
+
data.tar.gz: c0a9cb7f9f01e25dcce78f69894c23c8d5fe75d2d21fd20828d2daafa133186ebeb46d2aeb09bfc4246f43f851a3b982c4691cd013c40c56085634ca54f9db01
|
data/CHANGELOG.md
CHANGED
|
@@ -3,6 +3,42 @@
|
|
|
3
3
|
All notable changes to AsciiChem are documented here.
|
|
4
4
|
This project follows [Semantic Versioning](https://semver.org/).
|
|
5
5
|
|
|
6
|
+
## [0.22.0] - 2026-09-12
|
|
7
|
+
|
|
8
|
+
### Added
|
|
9
|
+
- CLI `--from asciichem|smiles|molfile` on `convert`, plus
|
|
10
|
+
`model-json`, `smiles`, `molfile`, and `structural-svg` output
|
|
11
|
+
targets (TODO.impl 60).
|
|
12
|
+
|
|
13
|
+
### Changed
|
|
14
|
+
- Conformance validation uses schemas vendored in `spec/schemas`
|
|
15
|
+
(json_schemer dev dependency); the asciichem-model rubygem
|
|
16
|
+
dependency is removed — the contract repository is no longer
|
|
17
|
+
distributed as a gem (TODO.impl 59).
|
|
18
|
+
|
|
19
|
+
## [0.21.0] - 2026-09-12
|
|
20
|
+
|
|
21
|
+
### Added
|
|
22
|
+
- Structure interchange (TODO.v2 09, TODO.impl 57): SMILES and
|
|
23
|
+
molfile (CTfile V2000) ingestion and emission as modules of the one
|
|
24
|
+
semantic model — `AsciiChem.parse_smiles` / `parse_molfile`,
|
|
25
|
+
`to_smiles` / `to_molfile`. Ingested molecules are ordinary
|
|
26
|
+
`Model::Molecule`s: graphs linearise into atoms + bond tokens +
|
|
27
|
+
ring-closure digits (`Structure::Linearizer`), so every existing
|
|
28
|
+
renderer, linter, and wire form works unchanged. The SMILES writer
|
|
29
|
+
is deterministic (DFS, single-bond continuations, order-independent
|
|
30
|
+
tie-breaks); aspirin and naphthalene round-trip exactly. v1
|
|
31
|
+
deferrals, each with an actionable `ParseError`: chirality `@`/`@@`,
|
|
32
|
+
E/Z directions `/` `\`, wildcard atoms, bonded ring closures.
|
|
33
|
+
- `Model::Atom#aromatic` / `#hydrogens` and an `aromatic` bond kind
|
|
34
|
+
(asciichem-model 0.4.0 fields): lowercase SMILES atoms, bracket
|
|
35
|
+
H-counts, molfile type-4 bonds (aromatic atoms marked from bonds).
|
|
36
|
+
- `AsciiChem::Structure` — shared graph walk + adjacency linearizer
|
|
37
|
+
for the interchange formats; `StructuralSvg` renders aromatic bonds
|
|
38
|
+
dashed; wire form carries the new fields both ways.
|
|
39
|
+
- Corpus levels: asciichem-tests v0.3.0 `structure/smiles/*` and
|
|
40
|
+
`structure/molfile/*` fixtures at 100% (37 + 6 cases).
|
|
41
|
+
|
|
6
42
|
## [0.20.0] - 2026-09-12
|
|
7
43
|
|
|
8
44
|
### Added
|
|
@@ -363,7 +399,9 @@ This project follows [Semantic Versioning](https://semver.org/).
|
|
|
363
399
|
`version`.
|
|
364
400
|
- Comprehensive RSpec suite with round-trip conformance.
|
|
365
401
|
|
|
366
|
-
[Unreleased]: https://github.com/asciichem/asciichem-ruby/compare/v0.
|
|
402
|
+
[Unreleased]: https://github.com/asciichem/asciichem-ruby/compare/v0.22.0...HEAD
|
|
403
|
+
[0.22.0]: https://github.com/asciichem/asciichem-ruby/compare/v0.21.0...v0.22.0
|
|
404
|
+
[0.21.0]: https://github.com/asciichem/asciichem-ruby/compare/v0.20.0...v0.21.0
|
|
367
405
|
[0.20.0]: https://github.com/asciichem/asciichem-ruby/compare/v0.19.0...v0.20.0
|
|
368
406
|
[0.18.1]: https://github.com/asciichem/asciichem-ruby/compare/v0.18.0...v0.18.1
|
|
369
407
|
[0.18.0]: https://github.com/asciichem/asciichem-ruby/compare/v0.17.0...v0.18.0
|
data/Gemfile
CHANGED
data/asciichem.gemspec
CHANGED
data/lib/asciichem/cli.rb
CHANGED
|
@@ -10,16 +10,22 @@ module AsciiChem
|
|
|
10
10
|
# and command banners, matching the executable name.
|
|
11
11
|
package_name "asciichem"
|
|
12
12
|
|
|
13
|
-
desc "convert -i INPUT -t FORMAT", "Convert
|
|
14
|
-
method_option :input, aliases: "-i", type: :string,
|
|
15
|
-
desc: "
|
|
13
|
+
desc "convert -i INPUT -t FORMAT", "Convert INPUT to FORMAT (mathml|text|html|latex|svg|structural-svg|model-json|cml|smiles|molfile)"
|
|
14
|
+
method_option :input, aliases: "-i", type: :string,
|
|
15
|
+
desc: "Source text (or '-' for stdin)"
|
|
16
16
|
method_option :file, aliases: "-f", type: :string,
|
|
17
|
-
desc: "Read
|
|
17
|
+
desc: "Read source from a file"
|
|
18
|
+
method_option :from, type: :string, default: "asciichem",
|
|
19
|
+
desc: "Input grammar: asciichem|smiles|molfile"
|
|
18
20
|
method_option :format, aliases: "-t", type: :string, default: "mathml",
|
|
19
21
|
desc: "Output format"
|
|
20
22
|
def convert
|
|
23
|
+
unless options["input"] || options["file"]
|
|
24
|
+
raise AsciiChem::ParseError, "provide -i INPUT or -f FILE"
|
|
25
|
+
end
|
|
26
|
+
|
|
21
27
|
source = read_source
|
|
22
|
-
formula =
|
|
28
|
+
formula = ingest(source, options[:from])
|
|
23
29
|
puts render(formula, options[:format])
|
|
24
30
|
rescue AsciiChem::ParseError => e
|
|
25
31
|
warn "Parse error: #{e.message}"
|
|
@@ -89,8 +95,31 @@ module AsciiChem
|
|
|
89
95
|
options[:input]
|
|
90
96
|
end
|
|
91
97
|
|
|
98
|
+
# One ingestion point per input grammar (TODO.v2 09): every
|
|
99
|
+
# grammar funnels into the same semantic model, so every output
|
|
100
|
+
# format works regardless of the input language.
|
|
101
|
+
def ingest(source, from)
|
|
102
|
+
case from.to_s
|
|
103
|
+
when "asciichem" then AsciiChem.parse(source)
|
|
104
|
+
when "smiles" then AsciiChem.parse_smiles(source)
|
|
105
|
+
when "molfile" then molfile_formula(source)
|
|
106
|
+
else
|
|
107
|
+
raise AsciiChem::ParseError, "unknown --from grammar: #{from}"
|
|
108
|
+
end
|
|
109
|
+
end
|
|
110
|
+
|
|
111
|
+
# parse_molfile returns a single Molecule; wrap it so every
|
|
112
|
+
# formatter's Formula contract holds.
|
|
113
|
+
def molfile_formula(source)
|
|
114
|
+
text = File.file?(source) ? File.read(source) : source
|
|
115
|
+
AsciiChem::Model::Formula.new(nodes: [AsciiChem.parse_molfile(text)])
|
|
116
|
+
end
|
|
117
|
+
|
|
92
118
|
def render(formula, format)
|
|
93
119
|
return formula.to_cml if format.to_sym == :cml
|
|
120
|
+
return formula.to_model_json if format.to_sym == :"model-json"
|
|
121
|
+
return formula.to_smiles if format.to_sym == :smiles
|
|
122
|
+
return formula.nodes.first.to_molfile if format.to_sym == :molfile
|
|
94
123
|
|
|
95
124
|
AsciiChem::Formatter.render(format.to_sym, formula)
|
|
96
125
|
end
|
|
@@ -258,6 +258,14 @@ module AsciiChem
|
|
|
258
258
|
RENDERERS = {
|
|
259
259
|
single: ->(r) { [r.base_line] },
|
|
260
260
|
|
|
261
|
+
# Aromatic bonds render as dashed lines (the inner-ring
|
|
262
|
+
# circle is a renderer nicety left for a later iteration).
|
|
263
|
+
aromatic: lambda do |r|
|
|
264
|
+
line = r.base_line
|
|
265
|
+
line['stroke-dasharray'] = '4 2.5'
|
|
266
|
+
[line]
|
|
267
|
+
end,
|
|
268
|
+
|
|
261
269
|
double: ->(r) { [-SPACING, 0, SPACING].map { |d| r.offset_line(d) }.compact },
|
|
262
270
|
|
|
263
271
|
triple: ->(r) { [0, -SPACING * 1.5, SPACING * 1.5].map { |d| r.offset_line(d) }.compact },
|
data/lib/asciichem/model/atom.rb
CHANGED
|
@@ -34,7 +34,7 @@ module AsciiChem
|
|
|
34
34
|
attr_accessor :element, :isotope, :subscript, :superscript,
|
|
35
35
|
:charge, :oxidation_state,
|
|
36
36
|
:lone_pairs, :radical_electrons,
|
|
37
|
-
:ring_closures,
|
|
37
|
+
:ring_closures, :aromatic, :hydrogens,
|
|
38
38
|
:x2, :y2, :z2, :atom_parity,
|
|
39
39
|
:spin_multiplicity, :atom_title,
|
|
40
40
|
:x_fract, :y_fract, :z_fract
|
|
@@ -58,7 +58,7 @@ module AsciiChem
|
|
|
58
58
|
def initialize(element:, isotope: nil, subscript: nil,
|
|
59
59
|
superscript: nil, charge: nil, oxidation_state: nil,
|
|
60
60
|
lone_pairs: nil, radical_electrons: nil,
|
|
61
|
-
ring_closures: nil,
|
|
61
|
+
ring_closures: nil, aromatic: nil, hydrogens: nil,
|
|
62
62
|
x2: nil, y2: nil, z2: nil, atom_parity: nil,
|
|
63
63
|
spin_multiplicity: nil, atom_title: nil,
|
|
64
64
|
x_fract: nil, y_fract: nil, z_fract: nil)
|
|
@@ -71,6 +71,8 @@ module AsciiChem
|
|
|
71
71
|
@lone_pairs = lone_pairs
|
|
72
72
|
@radical_electrons = radical_electrons
|
|
73
73
|
@ring_closures = ring_closures
|
|
74
|
+
@aromatic = aromatic
|
|
75
|
+
@hydrogens = hydrogens
|
|
74
76
|
@x2 = x2
|
|
75
77
|
@y2 = y2
|
|
76
78
|
@z2 = z2
|
|
@@ -87,7 +89,8 @@ module AsciiChem
|
|
|
87
89
|
superscript: superscript, charge: charge,
|
|
88
90
|
oxidation_state: oxidation_state,
|
|
89
91
|
lone_pairs: lone_pairs, radical_electrons: radical_electrons,
|
|
90
|
-
ring_closures: ring_closures,
|
|
92
|
+
ring_closures: ring_closures, aromatic: aromatic,
|
|
93
|
+
hydrogens: hydrogens,
|
|
91
94
|
x2: x2, y2: y2, z2: z2, atom_parity: atom_parity,
|
|
92
95
|
spin_multiplicity: spin_multiplicity, atom_title: atom_title,
|
|
93
96
|
x_fract: x_fract, y_fract: y_fract, z_fract: z_fract }
|
|
@@ -108,6 +111,8 @@ module AsciiChem
|
|
|
108
111
|
parts << "^(#{oxidation_state})" if oxidation_state
|
|
109
112
|
parts << ".#{radical_electrons}" if radical_electrons
|
|
110
113
|
parts << ring_closures.to_s if ring_closures
|
|
114
|
+
parts << "aromatic" if aromatic
|
|
115
|
+
parts << "H#{hydrogens}" if hydrogens
|
|
111
116
|
"Atom(#{parts.join})"
|
|
112
117
|
end
|
|
113
118
|
end
|
data/lib/asciichem/model/bond.rb
CHANGED
|
@@ -14,7 +14,8 @@ module AsciiChem
|
|
|
14
14
|
wedge: { ascii: ">-", mathml_entity: "↑" },
|
|
15
15
|
hash: { ascii: "-<", mathml_entity: "↓" },
|
|
16
16
|
dative: { ascii: "~>", mathml_entity: "→" },
|
|
17
|
-
wavy: { ascii: "~~", mathml_entity: "∼" }
|
|
17
|
+
wavy: { ascii: "~~", mathml_entity: "∼" },
|
|
18
|
+
aromatic: { ascii: ":", mathml_entity: ":" }
|
|
18
19
|
}.freeze
|
|
19
20
|
|
|
20
21
|
# CML wire order codes per bond kind. Single source of truth
|
data/lib/asciichem/model/node.rb
CHANGED
|
@@ -70,6 +70,18 @@ module AsciiChem
|
|
|
70
70
|
AsciiChem::WireAdapter.to_model_json(self)
|
|
71
71
|
end
|
|
72
72
|
|
|
73
|
+
# Deterministic SMILES for a Molecule (or dot-joined components
|
|
74
|
+
# for a Formula). Raises for constructs with no SMILES form.
|
|
75
|
+
def to_smiles
|
|
76
|
+
AsciiChem::Smiles.write(self)
|
|
77
|
+
end
|
|
78
|
+
|
|
79
|
+
# Molfile V2000 for a Molecule. Authored coordinates win; a
|
|
80
|
+
# deterministic 2D layout is computed otherwise.
|
|
81
|
+
def to_molfile(name: nil)
|
|
82
|
+
AsciiChem::Molfile.write(self, name: name)
|
|
83
|
+
end
|
|
84
|
+
|
|
73
85
|
# Subclasses override to expose the attributes that participate in
|
|
74
86
|
# equality. Default: empty (so two bare Nodes are equal).
|
|
75
87
|
def value_attributes
|
|
@@ -0,0 +1,169 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module AsciiChem
|
|
4
|
+
module Molfile
|
|
5
|
+
# V2000 molfile → Model::Molecule. Fixed-format blocks: header
|
|
6
|
+
# (3 lines), counts line, atom block, bond block, property block.
|
|
7
|
+
# Charges come from `M CHG`, isotopes from the mass-difference
|
|
8
|
+
# field or `M ISO`, bond stereo codes 1/6 become wedge/hash.
|
|
9
|
+
class Parser
|
|
10
|
+
def initialize(text)
|
|
11
|
+
@lines = text.lines.map(&:chomp)
|
|
12
|
+
end
|
|
13
|
+
|
|
14
|
+
def parse
|
|
15
|
+
raise ParseError, "molfile too short" if @lines.length < 5
|
|
16
|
+
|
|
17
|
+
atom_field = field(3, 0, 3)
|
|
18
|
+
bond_field = field(3, 3, 3)
|
|
19
|
+
unless atom_field.match?(/\A\d+\z/) && bond_field.match?(/\A\d+\z/)
|
|
20
|
+
raise ParseError, "malformed counts line: #{@lines[3].inspect}"
|
|
21
|
+
end
|
|
22
|
+
atom_count = atom_field.to_i
|
|
23
|
+
bond_count = bond_field.to_i
|
|
24
|
+
|
|
25
|
+
atoms = parse_atoms(atom_count)
|
|
26
|
+
bonds = parse_bonds(bond_count)
|
|
27
|
+
properties = parse_properties
|
|
28
|
+
|
|
29
|
+
apply_legacy_charge!(atoms)
|
|
30
|
+
apply_charges!(atoms, properties[:charges])
|
|
31
|
+
apply_isotopes!(atoms, properties[:isotopes])
|
|
32
|
+
|
|
33
|
+
adjacency = Array.new(atoms.length) { {} }
|
|
34
|
+
bonds.each do |bond|
|
|
35
|
+
adjacency[bond[:from]][bond[:to]] = bond[:kind]
|
|
36
|
+
adjacency[bond[:to]][bond[:from]] = bond[:kind]
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
# V2000 carries aromaticity on bonds (type 4); the model
|
|
40
|
+
# carries it on atoms and bonds, so atoms touching an
|
|
41
|
+
# aromatic bond are marked aromatic.
|
|
42
|
+
bonds.each do |bond|
|
|
43
|
+
next unless bond[:kind] == :aromatic
|
|
44
|
+
|
|
45
|
+
atoms[bond[:from]].aromatic = true
|
|
46
|
+
atoms[bond[:to]].aromatic = true
|
|
47
|
+
end
|
|
48
|
+
|
|
49
|
+
Model::Molecule.new(
|
|
50
|
+
nodes: Structure::Linearizer.new(atoms: atoms, edges: adjacency_to_edges(adjacency)).nodes
|
|
51
|
+
)
|
|
52
|
+
end
|
|
53
|
+
|
|
54
|
+
private
|
|
55
|
+
|
|
56
|
+
def field(line_index, start, length)
|
|
57
|
+
(@lines[line_index] || "")[start, length].to_s.strip
|
|
58
|
+
end
|
|
59
|
+
|
|
60
|
+
def parse_atoms(count)
|
|
61
|
+
(1..count).map do |i|
|
|
62
|
+
line_index = 3 + i
|
|
63
|
+
line = @lines[line_index]
|
|
64
|
+
raise ParseError, "truncated atom block (expected #{count} atoms)" if line.nil?
|
|
65
|
+
|
|
66
|
+
x = line[0, 10].to_f
|
|
67
|
+
y = line[10, 10].to_f
|
|
68
|
+
z = line[20, 10].to_f
|
|
69
|
+
element = line[31, 3].to_s.strip
|
|
70
|
+
mass_diff = line[34, 2].to_i
|
|
71
|
+
raise ParseError, "atom #{i} has no element symbol" if element.empty?
|
|
72
|
+
|
|
73
|
+
Model::Atom.new(
|
|
74
|
+
element: element,
|
|
75
|
+
x2: x, y2: y, z2: z,
|
|
76
|
+
isotope: isotope_from_mass_diff(element, mass_diff)
|
|
77
|
+
)
|
|
78
|
+
end
|
|
79
|
+
end
|
|
80
|
+
|
|
81
|
+
def parse_bonds(count)
|
|
82
|
+
(1..count).map do |i|
|
|
83
|
+
line = @lines[3 + atoms_count + i]
|
|
84
|
+
raise ParseError, "truncated bond block (expected #{count} bonds)" if line.nil?
|
|
85
|
+
|
|
86
|
+
from = line[0, 3].to_i - 1
|
|
87
|
+
to = line[3, 3].to_i - 1
|
|
88
|
+
type = line[6, 3].to_i
|
|
89
|
+
stereo = line[9, 3].to_i
|
|
90
|
+
raise ParseError, "bond #{i} has out-of-range atom indexes" if from.negative? || to.negative?
|
|
91
|
+
|
|
92
|
+
{ from: from, to: to, kind: bond_kind(type, stereo) }
|
|
93
|
+
end
|
|
94
|
+
end
|
|
95
|
+
|
|
96
|
+
def parse_properties
|
|
97
|
+
charges = {}
|
|
98
|
+
isotopes = {}
|
|
99
|
+
@lines.each do |line|
|
|
100
|
+
if line.start_with?("M CHG")
|
|
101
|
+
parts = line[6..].split
|
|
102
|
+
_count = parts[0].to_i
|
|
103
|
+
parts[1..].each_slice(2) do |idx, charge|
|
|
104
|
+
charges[idx.to_i - 1] = charge.to_i if idx && charge
|
|
105
|
+
end
|
|
106
|
+
elsif line.start_with?("M ISO")
|
|
107
|
+
parts = line[6..].split
|
|
108
|
+
parts[1..].each_slice(2) do |idx, mass|
|
|
109
|
+
isotopes[idx.to_i - 1] = mass.to_i if idx && mass
|
|
110
|
+
end
|
|
111
|
+
end
|
|
112
|
+
end
|
|
113
|
+
{ charges: charges, isotopes: isotopes }
|
|
114
|
+
end
|
|
115
|
+
|
|
116
|
+
def bond_kind(type, stereo)
|
|
117
|
+
return :wedge if stereo == 1
|
|
118
|
+
return :hash if stereo == 6
|
|
119
|
+
|
|
120
|
+
{ 1 => :single, 2 => :double, 3 => :triple, 4 => :aromatic }[type] ||
|
|
121
|
+
raise(ParseError, "unsupported molfile bond type #{type}")
|
|
122
|
+
end
|
|
123
|
+
|
|
124
|
+
# Pre-CHG charge column (0-based 36, 3): nonzero values 1..4
|
|
125
|
+
# mean +1..+4, 5..7 mean -1..-3. `M CHG` overrides.
|
|
126
|
+
def apply_legacy_charge!(atoms)
|
|
127
|
+
(1..field(3, 0, 3).to_i).each do |i|
|
|
128
|
+
code = field(3 + i, 36, 3).to_i
|
|
129
|
+
next if code.zero?
|
|
130
|
+
|
|
131
|
+
charge = code <= 4 ? code : 4 - code
|
|
132
|
+
sign = charge.negative? ? "-" : "+"
|
|
133
|
+
atoms[i - 1].charge = charge.abs == 1 ? sign : "#{charge.abs}#{sign}"
|
|
134
|
+
end
|
|
135
|
+
end
|
|
136
|
+
|
|
137
|
+
def apply_charges!(atoms, charges)
|
|
138
|
+
charges.each do |index, value|
|
|
139
|
+
sign = value.negative? ? "-" : "+"
|
|
140
|
+
atoms[index].charge = value.abs == 1 ? sign : "#{value.abs}#{sign}"
|
|
141
|
+
end
|
|
142
|
+
end
|
|
143
|
+
|
|
144
|
+
def apply_isotopes!(atoms, isotopes)
|
|
145
|
+
isotopes.each do |index, mass|
|
|
146
|
+
atoms[index].isotope = mass.to_s
|
|
147
|
+
end
|
|
148
|
+
end
|
|
149
|
+
|
|
150
|
+
# Mass difference encodes isotopes relative to the rounded
|
|
151
|
+
# average mass; mapping it unambiguously requires isotope
|
|
152
|
+
# tables, so v1 defers isotopes to the explicit `M ISO` block.
|
|
153
|
+
def isotope_from_mass_diff(_element, mass_diff)
|
|
154
|
+
nil
|
|
155
|
+
end
|
|
156
|
+
def atoms_count
|
|
157
|
+
field(3, 0, 3).to_i
|
|
158
|
+
end
|
|
159
|
+
|
|
160
|
+
def adjacency_to_edges(adjacency)
|
|
161
|
+
edges = []
|
|
162
|
+
adjacency.each_with_index do |neighbors, index|
|
|
163
|
+
neighbors.each { |to, kind| edges << Structure::Graph::Edge.new(from: index, to: to, kind: kind) if to > index }
|
|
164
|
+
end
|
|
165
|
+
edges
|
|
166
|
+
end
|
|
167
|
+
end
|
|
168
|
+
end
|
|
169
|
+
end
|
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module AsciiChem
|
|
4
|
+
module Molfile
|
|
5
|
+
# Model molecule → V2000 molfile. Authored x2/y2 coordinates are
|
|
6
|
+
# used when present; otherwise a deterministic 2D layout is
|
|
7
|
+
# computed (Layout walks atoms in the same order as
|
|
8
|
+
# Structure::Graph, so positions map by index). Charges and
|
|
9
|
+
# isotopes are emitted as `M CHG` / `M ISO` property lines.
|
|
10
|
+
class Writer
|
|
11
|
+
BOND_TYPES = {
|
|
12
|
+
single: 1, double: 2, triple: 3, aromatic: 4,
|
|
13
|
+
wedge: 1, hash: 1
|
|
14
|
+
}.freeze
|
|
15
|
+
BOND_STEREO = { wedge: 1, hash: 6 }.freeze
|
|
16
|
+
|
|
17
|
+
def initialize(molecule, name: nil)
|
|
18
|
+
@molecule = molecule
|
|
19
|
+
@name = name
|
|
20
|
+
end
|
|
21
|
+
|
|
22
|
+
def write
|
|
23
|
+
atoms, edges = Structure::Graph.build(@molecule)
|
|
24
|
+
raise ParseError, "molecule has no bonds — a formula is not a structure" if edges.empty?
|
|
25
|
+
|
|
26
|
+
lines = []
|
|
27
|
+
lines << @name.to_s
|
|
28
|
+
lines << " AsciiChem"
|
|
29
|
+
lines << ""
|
|
30
|
+
lines << format("%3d%3d 0 0 0 0 0 0 0 0999 V2000", atoms.length, edges.length)
|
|
31
|
+
|
|
32
|
+
layout = nil
|
|
33
|
+
atoms.each_with_index do |atom, index|
|
|
34
|
+
x, y = coordinates(atom, index, atoms)
|
|
35
|
+
lines << format("%10.4f%10.4f%10.4f %-3s 0 0 0 0 0 0 0 0 0 0 0 0",
|
|
36
|
+
x, y, atom.z2 || 0.0, atom.element)
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
edges.each do |edge|
|
|
40
|
+
type = BOND_TYPES.fetch(edge.kind) do
|
|
41
|
+
raise ParseError, "#{edge.kind} bonds have no molfile V2000 type"
|
|
42
|
+
end
|
|
43
|
+
stereo = BOND_STEREO.fetch(edge.kind, 0)
|
|
44
|
+
lines << format("%3d%3d%3d%3d 0 0 0 0 0 0 0",
|
|
45
|
+
edge.from + 1, edge.to + 1, type, stereo)
|
|
46
|
+
end
|
|
47
|
+
|
|
48
|
+
lines << property_line("M CHG", charge_pairs(atoms))
|
|
49
|
+
lines << property_line("M ISO", isotope_pairs(atoms))
|
|
50
|
+
lines << "M END"
|
|
51
|
+
lines.compact.join("\n") << "\n"
|
|
52
|
+
end
|
|
53
|
+
|
|
54
|
+
private
|
|
55
|
+
|
|
56
|
+
# Authored coordinates win; otherwise positions from the
|
|
57
|
+
# deterministic 2D layout (same walk order as the graph).
|
|
58
|
+
def coordinates(atom, index, atoms)
|
|
59
|
+
return [atom.x2, atom.y2] if atom.x2 && atom.y2
|
|
60
|
+
|
|
61
|
+
@layout ||= AsciiChem::Layout.layout(@molecule)
|
|
62
|
+
placed = @layout.atoms[index]
|
|
63
|
+
placed ? [placed.x, placed.y] : [0.0, 0.0]
|
|
64
|
+
end
|
|
65
|
+
|
|
66
|
+
def charge_pairs(atoms)
|
|
67
|
+
atoms.each_with_index
|
|
68
|
+
.filter_map { |atom, i| [i + 1, charge_value(atom.charge)] if atom.charge }
|
|
69
|
+
end
|
|
70
|
+
|
|
71
|
+
def isotope_pairs(atoms)
|
|
72
|
+
atoms.each_with_index
|
|
73
|
+
.filter_map { |atom, i| [i + 1, atom.isotope.to_i] if atom.isotope }
|
|
74
|
+
end
|
|
75
|
+
|
|
76
|
+
# The model's number-then-sign charge ("2+") → signed integer.
|
|
77
|
+
def charge_value(charge)
|
|
78
|
+
count = charge[/\A\d+/]&.to_i || 1
|
|
79
|
+
charge.end_with?("-") ? -count : count
|
|
80
|
+
end
|
|
81
|
+
|
|
82
|
+
def property_line(prefix, pairs)
|
|
83
|
+
return nil if pairs.empty?
|
|
84
|
+
|
|
85
|
+
pairs.reduce(+"#{prefix}%3d" % pairs.length) do |line, (idx, value)|
|
|
86
|
+
line << format("%4d%4d", idx, value)
|
|
87
|
+
end
|
|
88
|
+
end
|
|
89
|
+
end
|
|
90
|
+
end
|
|
91
|
+
end
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module AsciiChem
|
|
4
|
+
# Molfile (CTfile V2000) ingestion and emission (TODO.v2 09;
|
|
5
|
+
# TODO.impl 57). Molfile is the highest-fidelity structure path:
|
|
6
|
+
# atom coordinates are preserved on the model (x2/y2/z2), charges
|
|
7
|
+
# via the CHG property block, isotopes via mass difference or the
|
|
8
|
+
# ISO block, bond stereo codes 1/6 map to wedge/hash bonds.
|
|
9
|
+
module Molfile
|
|
10
|
+
autoload :Parser, "asciichem/molfile/parser"
|
|
11
|
+
autoload :Writer, "asciichem/molfile/writer"
|
|
12
|
+
|
|
13
|
+
class << self
|
|
14
|
+
# Parses a V2000 molfile into a Model::Molecule.
|
|
15
|
+
def parse(text)
|
|
16
|
+
Parser.new(text).parse
|
|
17
|
+
end
|
|
18
|
+
|
|
19
|
+
# Emits a V2000 molfile. Uses authored x2/y2 coordinates;
|
|
20
|
+
# computes a 2D layout otherwise.
|
|
21
|
+
def write(molecule, name: nil)
|
|
22
|
+
Writer.new(molecule, name: name).write
|
|
23
|
+
end
|
|
24
|
+
end
|
|
25
|
+
end
|
|
26
|
+
end
|