asciichem 0.19.0 → 0.21.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.github/workflows/ci.yml +9 -0
- data/.github/workflows/release.yml +7 -17
- data/.gitignore +1 -0
- data/CHANGELOG.md +42 -1
- data/RELEASING.md +9 -4
- data/asciichem.gemspec +3 -0
- data/lib/asciichem/formatter/structural_svg.rb +8 -0
- data/lib/asciichem/model/atom.rb +8 -3
- data/lib/asciichem/model/bond.rb +2 -1
- data/lib/asciichem/model/node.rb +20 -0
- data/lib/asciichem/molfile/parser.rb +169 -0
- data/lib/asciichem/molfile/writer.rb +91 -0
- data/lib/asciichem/molfile.rb +26 -0
- data/lib/asciichem/smiles/parser.rb +302 -0
- data/lib/asciichem/smiles/writer.rb +210 -0
- data/lib/asciichem/smiles.rb +45 -0
- data/lib/asciichem/structure/graph.rb +62 -0
- data/lib/asciichem/structure/linearizer.rb +67 -0
- data/lib/asciichem/structure.rb +20 -0
- data/lib/asciichem/version.rb +1 -1
- data/lib/asciichem/wire/base.rb +25 -0
- data/lib/asciichem/wire/chemistry.rb +98 -0
- data/lib/asciichem/wire/core.rb +96 -0
- data/lib/asciichem/wire/extended.rb +136 -0
- data/lib/asciichem/wire/identity.rb +63 -0
- data/lib/asciichem/wire.rb +52 -0
- data/lib/asciichem/wire_adapter.rb +368 -0
- data/lib/asciichem.rb +25 -0
- metadata +52 -2
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: 2dee0024c277879fed3b5695a76e336067660467724663c4a67f71ec87215e80
|
|
4
|
+
data.tar.gz: 50d5a53fd8d4f64a0d3969c586e81552faed9221a8f209932cd6823a2a966327
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: 3afdd1db8e99729ca9a1a49be7a748fd56d1fdb666b5bfe10bb03a08c6d7db440ca82b3c900b6f22f09ac66901d7c086913a2a3e8c142bc6c7d776f21959c629
|
|
7
|
+
data.tar.gz: ba3091db4ea3a3d7ba8dcd28f9689df323b135e1e1e4f0a3f3573f14350448461f9f3ade4e20fe4e7af9408d0ce40c1764bba4d609a05bd770705be431ca497b
|
data/.github/workflows/ci.yml
CHANGED
|
@@ -14,8 +14,17 @@ jobs:
|
|
|
14
14
|
ruby: ["3.3", "3.4"]
|
|
15
15
|
steps:
|
|
16
16
|
- uses: actions/checkout@v4
|
|
17
|
+
- name: Clone conformance corpus (asciichem-tests)
|
|
18
|
+
run: git clone --depth 1 https://github.com/asciichem/asciichem-tests.git ../asciichem-tests
|
|
17
19
|
- uses: ruby/setup-ruby@v1
|
|
18
20
|
with:
|
|
19
21
|
ruby-version: ${{ matrix.ruby }}
|
|
20
22
|
bundler-cache: true
|
|
21
23
|
- run: bundle exec rspec
|
|
24
|
+
- name: Publish conformance report
|
|
25
|
+
if: matrix.ruby == '3.4'
|
|
26
|
+
uses: actions/upload-artifact@v4
|
|
27
|
+
with:
|
|
28
|
+
name: conformance
|
|
29
|
+
path: conformance.json
|
|
30
|
+
if-no-files-found: warn
|
|
@@ -11,14 +11,17 @@ on:
|
|
|
11
11
|
jobs:
|
|
12
12
|
release:
|
|
13
13
|
runs-on: ubuntu-latest
|
|
14
|
-
|
|
14
|
+
# Trusted publishing (OIDC): the publisher registered on RubyGems.org
|
|
15
|
+
# is repository asciichem/asciichem-ruby + workflow release.yml, with
|
|
16
|
+
# no environment — so this job must not claim one.
|
|
15
17
|
permissions:
|
|
16
18
|
contents: read
|
|
17
|
-
|
|
19
|
+
id-token: write
|
|
18
20
|
steps:
|
|
19
21
|
- uses: actions/checkout@v4
|
|
20
22
|
with:
|
|
21
23
|
ref: main
|
|
24
|
+
persist-credentials: false
|
|
22
25
|
- name: Verify version matches input
|
|
23
26
|
run: |
|
|
24
27
|
actual=$(ruby -e 'require "./lib/asciichem/version"; print AsciiChem::VERSION')
|
|
@@ -31,21 +34,8 @@ jobs:
|
|
|
31
34
|
with:
|
|
32
35
|
ruby-version: "3.4"
|
|
33
36
|
bundler-cache: true
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
- name: Set up RubyGems credentials
|
|
37
|
-
env:
|
|
38
|
-
RUBYGEMS_API_KEY: ${{ secrets.RUBYGEMS_API_KEY }}
|
|
39
|
-
run: |
|
|
40
|
-
mkdir -p $HOME/.gem
|
|
41
|
-
cat > $HOME/.gem/credentials <<EOF
|
|
42
|
-
---
|
|
43
|
-
:rubygems_api_key: ${RUBYGEMS_API_KEY}
|
|
44
|
-
EOF
|
|
45
|
-
chmod 0600 $HOME/.gem/credentials
|
|
46
|
-
- name: Push to RubyGems
|
|
47
|
-
run: |
|
|
48
|
-
gem push pkg/asciichem-${{ inputs.version }}.gem
|
|
37
|
+
# Builds and pushes using the GitHub OIDC identity — no API keys.
|
|
38
|
+
- uses: rubygems/release-gem@v1
|
|
49
39
|
- name: Summary
|
|
50
40
|
run: |
|
|
51
41
|
echo "Released asciichem ${{ inputs.version }} to RubyGems"
|
data/.gitignore
CHANGED
data/CHANGELOG.md
CHANGED
|
@@ -3,6 +3,45 @@
|
|
|
3
3
|
All notable changes to AsciiChem are documented here.
|
|
4
4
|
This project follows [Semantic Versioning](https://semver.org/).
|
|
5
5
|
|
|
6
|
+
## [0.21.0] - 2026-09-12
|
|
7
|
+
|
|
8
|
+
### Added
|
|
9
|
+
- Structure interchange (TODO.v2 09, TODO.impl 57): SMILES and
|
|
10
|
+
molfile (CTfile V2000) ingestion and emission as modules of the one
|
|
11
|
+
semantic model — `AsciiChem.parse_smiles` / `parse_molfile`,
|
|
12
|
+
`to_smiles` / `to_molfile`. Ingested molecules are ordinary
|
|
13
|
+
`Model::Molecule`s: graphs linearise into atoms + bond tokens +
|
|
14
|
+
ring-closure digits (`Structure::Linearizer`), so every existing
|
|
15
|
+
renderer, linter, and wire form works unchanged. The SMILES writer
|
|
16
|
+
is deterministic (DFS, single-bond continuations, order-independent
|
|
17
|
+
tie-breaks); aspirin and naphthalene round-trip exactly. v1
|
|
18
|
+
deferrals, each with an actionable `ParseError`: chirality `@`/`@@`,
|
|
19
|
+
E/Z directions `/` `\`, wildcard atoms, bonded ring closures.
|
|
20
|
+
- `Model::Atom#aromatic` / `#hydrogens` and an `aromatic` bond kind
|
|
21
|
+
(asciichem-model 0.4.0 fields): lowercase SMILES atoms, bracket
|
|
22
|
+
H-counts, molfile type-4 bonds (aromatic atoms marked from bonds).
|
|
23
|
+
- `AsciiChem::Structure` — shared graph walk + adjacency linearizer
|
|
24
|
+
for the interchange formats; `StructuralSvg` renders aromatic bonds
|
|
25
|
+
dashed; wire form carries the new fields both ways.
|
|
26
|
+
- Corpus levels: asciichem-tests v0.3.0 `structure/smiles/*` and
|
|
27
|
+
`structure/molfile/*` fixtures at 100% (37 + 6 cases).
|
|
28
|
+
|
|
29
|
+
## [0.20.0] - 2026-09-12
|
|
30
|
+
|
|
31
|
+
### Added
|
|
32
|
+
- Canonical JSON wire form (`to_model_json` / `AsciiChem.from_model_json`)
|
|
33
|
+
per asciichem-model v1: `AsciiChem::Wire` (lutaml-model Serializable
|
|
34
|
+
classes, json mappings only - no hand-rolled serialization) bridged by
|
|
35
|
+
`AsciiChem::WireAdapter` (model-to-model conversion, same pattern as
|
|
36
|
+
the CML ModelAdapter). Emission covers every node type; ingestion
|
|
37
|
+
covers the lossless core set (beyond-formulas nodes are emission-only
|
|
38
|
+
until their corpus round-trip acceptance lands).
|
|
39
|
+
- Conformance runner over the shared corpus (asciichem-tests): L0
|
|
40
|
+
emission + schema validation, L1 Text round-trip, L3 CML round-trip,
|
|
41
|
+
L4 linter diagnostics, plus the ParseError contract for rejects.
|
|
42
|
+
Emits conformance.json; CI clones the corpus and publishes the report
|
|
43
|
+
as an artifact. Current claim: L0 149/149, L1 16/16, L3 23/23, L4 6/6.
|
|
44
|
+
|
|
6
45
|
## [0.19.0] - 2026-09-09
|
|
7
46
|
|
|
8
47
|
### Added
|
|
@@ -347,7 +386,9 @@ This project follows [Semantic Versioning](https://semver.org/).
|
|
|
347
386
|
`version`.
|
|
348
387
|
- Comprehensive RSpec suite with round-trip conformance.
|
|
349
388
|
|
|
350
|
-
[Unreleased]: https://github.com/asciichem/asciichem-ruby/
|
|
389
|
+
[Unreleased]: https://github.com/asciichem/asciichem-ruby/compare/v0.21.0...HEAD
|
|
390
|
+
[0.21.0]: https://github.com/asciichem/asciichem-ruby/compare/v0.20.0...v0.21.0
|
|
391
|
+
[0.20.0]: https://github.com/asciichem/asciichem-ruby/compare/v0.19.0...v0.20.0
|
|
351
392
|
[0.18.1]: https://github.com/asciichem/asciichem-ruby/compare/v0.18.0...v0.18.1
|
|
352
393
|
[0.18.0]: https://github.com/asciichem/asciichem-ruby/compare/v0.17.0...v0.18.0
|
|
353
394
|
[0.17.0]: https://github.com/asciichem/asciichem-ruby/compare/v0.16.0...v0.17.0
|
data/RELEASING.md
CHANGED
|
@@ -83,10 +83,15 @@ tag.
|
|
|
83
83
|
|
|
84
84
|
== Publish to RubyGems
|
|
85
85
|
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
86
|
+
The Release workflow (`.github/workflows/release.yml`) publishes via
|
|
87
|
+
https://guides.rubygems.org/trusted-publishing/[RubyGems trusted publishing]
|
|
88
|
+
(OIDC) — there are no API keys. Its trusted publisher is registered on
|
|
89
|
+
https://rubygems.org (repository `asciichem/asciichem-ruby`, workflow
|
|
90
|
+
`release.yml`, no environment).
|
|
91
|
+
|
|
92
|
+
On the repository's Actions tab, run the "Release" workflow with the
|
|
93
|
+
version number. It verifies the version matches `version.rb`, builds
|
|
94
|
+
the gem, and pushes it using the workflow's GitHub OIDC identity.
|
|
90
95
|
|
|
91
96
|
Verify at https://rubygems.org/gems/asciichem.
|
|
92
97
|
|
data/asciichem.gemspec
CHANGED
|
@@ -35,8 +35,11 @@ Gem::Specification.new do |spec|
|
|
|
35
35
|
|
|
36
36
|
spec.add_dependency "chemicalml", "~> 0.3.0"
|
|
37
37
|
spec.add_dependency "elkrb", "~> 1.0"
|
|
38
|
+
spec.add_dependency "lutaml-model", ">= 0.8", "< 2"
|
|
38
39
|
spec.add_dependency "nokogiri", "~> 1.16"
|
|
39
40
|
spec.add_dependency "parslet", "~> 2.0"
|
|
40
41
|
spec.add_dependency "plurimath", "~> 0.8"
|
|
41
42
|
spec.add_dependency "thor", "~> 1.3"
|
|
43
|
+
|
|
44
|
+
spec.add_development_dependency "asciichem-model", "~> 0.3.4"
|
|
42
45
|
end
|
|
@@ -258,6 +258,14 @@ module AsciiChem
|
|
|
258
258
|
RENDERERS = {
|
|
259
259
|
single: ->(r) { [r.base_line] },
|
|
260
260
|
|
|
261
|
+
# Aromatic bonds render as dashed lines (the inner-ring
|
|
262
|
+
# circle is a renderer nicety left for a later iteration).
|
|
263
|
+
aromatic: lambda do |r|
|
|
264
|
+
line = r.base_line
|
|
265
|
+
line['stroke-dasharray'] = '4 2.5'
|
|
266
|
+
[line]
|
|
267
|
+
end,
|
|
268
|
+
|
|
261
269
|
double: ->(r) { [-SPACING, 0, SPACING].map { |d| r.offset_line(d) }.compact },
|
|
262
270
|
|
|
263
271
|
triple: ->(r) { [0, -SPACING * 1.5, SPACING * 1.5].map { |d| r.offset_line(d) }.compact },
|
data/lib/asciichem/model/atom.rb
CHANGED
|
@@ -34,7 +34,7 @@ module AsciiChem
|
|
|
34
34
|
attr_accessor :element, :isotope, :subscript, :superscript,
|
|
35
35
|
:charge, :oxidation_state,
|
|
36
36
|
:lone_pairs, :radical_electrons,
|
|
37
|
-
:ring_closures,
|
|
37
|
+
:ring_closures, :aromatic, :hydrogens,
|
|
38
38
|
:x2, :y2, :z2, :atom_parity,
|
|
39
39
|
:spin_multiplicity, :atom_title,
|
|
40
40
|
:x_fract, :y_fract, :z_fract
|
|
@@ -58,7 +58,7 @@ module AsciiChem
|
|
|
58
58
|
def initialize(element:, isotope: nil, subscript: nil,
|
|
59
59
|
superscript: nil, charge: nil, oxidation_state: nil,
|
|
60
60
|
lone_pairs: nil, radical_electrons: nil,
|
|
61
|
-
ring_closures: nil,
|
|
61
|
+
ring_closures: nil, aromatic: nil, hydrogens: nil,
|
|
62
62
|
x2: nil, y2: nil, z2: nil, atom_parity: nil,
|
|
63
63
|
spin_multiplicity: nil, atom_title: nil,
|
|
64
64
|
x_fract: nil, y_fract: nil, z_fract: nil)
|
|
@@ -71,6 +71,8 @@ module AsciiChem
|
|
|
71
71
|
@lone_pairs = lone_pairs
|
|
72
72
|
@radical_electrons = radical_electrons
|
|
73
73
|
@ring_closures = ring_closures
|
|
74
|
+
@aromatic = aromatic
|
|
75
|
+
@hydrogens = hydrogens
|
|
74
76
|
@x2 = x2
|
|
75
77
|
@y2 = y2
|
|
76
78
|
@z2 = z2
|
|
@@ -87,7 +89,8 @@ module AsciiChem
|
|
|
87
89
|
superscript: superscript, charge: charge,
|
|
88
90
|
oxidation_state: oxidation_state,
|
|
89
91
|
lone_pairs: lone_pairs, radical_electrons: radical_electrons,
|
|
90
|
-
ring_closures: ring_closures,
|
|
92
|
+
ring_closures: ring_closures, aromatic: aromatic,
|
|
93
|
+
hydrogens: hydrogens,
|
|
91
94
|
x2: x2, y2: y2, z2: z2, atom_parity: atom_parity,
|
|
92
95
|
spin_multiplicity: spin_multiplicity, atom_title: atom_title,
|
|
93
96
|
x_fract: x_fract, y_fract: y_fract, z_fract: z_fract }
|
|
@@ -108,6 +111,8 @@ module AsciiChem
|
|
|
108
111
|
parts << "^(#{oxidation_state})" if oxidation_state
|
|
109
112
|
parts << ".#{radical_electrons}" if radical_electrons
|
|
110
113
|
parts << ring_closures.to_s if ring_closures
|
|
114
|
+
parts << "aromatic" if aromatic
|
|
115
|
+
parts << "H#{hydrogens}" if hydrogens
|
|
111
116
|
"Atom(#{parts.join})"
|
|
112
117
|
end
|
|
113
118
|
end
|
data/lib/asciichem/model/bond.rb
CHANGED
|
@@ -14,7 +14,8 @@ module AsciiChem
|
|
|
14
14
|
wedge: { ascii: ">-", mathml_entity: "↑" },
|
|
15
15
|
hash: { ascii: "-<", mathml_entity: "↓" },
|
|
16
16
|
dative: { ascii: "~>", mathml_entity: "→" },
|
|
17
|
-
wavy: { ascii: "~~", mathml_entity: "∼" }
|
|
17
|
+
wavy: { ascii: "~~", mathml_entity: "∼" },
|
|
18
|
+
aromatic: { ascii: ":", mathml_entity: ":" }
|
|
18
19
|
}.freeze
|
|
19
20
|
|
|
20
21
|
# CML wire order codes per bond kind. Single source of truth
|
data/lib/asciichem/model/node.rb
CHANGED
|
@@ -62,6 +62,26 @@ module AsciiChem
|
|
|
62
62
|
AsciiChem::Cml.from_asciichem(self)
|
|
63
63
|
end
|
|
64
64
|
|
|
65
|
+
# Canonical JSON wire form (asciichem-model v1). The Text
|
|
66
|
+
# formatter canonicalises AsciiChem text; this canonicalises
|
|
67
|
+
# the semantic model itself — the interchange format every
|
|
68
|
+
# implementation must parse and emit.
|
|
69
|
+
def to_model_json
|
|
70
|
+
AsciiChem::WireAdapter.to_model_json(self)
|
|
71
|
+
end
|
|
72
|
+
|
|
73
|
+
# Deterministic SMILES for a Molecule (or dot-joined components
|
|
74
|
+
# for a Formula). Raises for constructs with no SMILES form.
|
|
75
|
+
def to_smiles
|
|
76
|
+
AsciiChem::Smiles.write(self)
|
|
77
|
+
end
|
|
78
|
+
|
|
79
|
+
# Molfile V2000 for a Molecule. Authored coordinates win; a
|
|
80
|
+
# deterministic 2D layout is computed otherwise.
|
|
81
|
+
def to_molfile(name: nil)
|
|
82
|
+
AsciiChem::Molfile.write(self, name: name)
|
|
83
|
+
end
|
|
84
|
+
|
|
65
85
|
# Subclasses override to expose the attributes that participate in
|
|
66
86
|
# equality. Default: empty (so two bare Nodes are equal).
|
|
67
87
|
def value_attributes
|
|
@@ -0,0 +1,169 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module AsciiChem
|
|
4
|
+
module Molfile
|
|
5
|
+
# V2000 molfile → Model::Molecule. Fixed-format blocks: header
|
|
6
|
+
# (3 lines), counts line, atom block, bond block, property block.
|
|
7
|
+
# Charges come from `M CHG`, isotopes from the mass-difference
|
|
8
|
+
# field or `M ISO`, bond stereo codes 1/6 become wedge/hash.
|
|
9
|
+
class Parser
|
|
10
|
+
def initialize(text)
|
|
11
|
+
@lines = text.lines.map(&:chomp)
|
|
12
|
+
end
|
|
13
|
+
|
|
14
|
+
def parse
|
|
15
|
+
raise ParseError, "molfile too short" if @lines.length < 5
|
|
16
|
+
|
|
17
|
+
atom_field = field(3, 0, 3)
|
|
18
|
+
bond_field = field(3, 3, 3)
|
|
19
|
+
unless atom_field.match?(/\A\d+\z/) && bond_field.match?(/\A\d+\z/)
|
|
20
|
+
raise ParseError, "malformed counts line: #{@lines[3].inspect}"
|
|
21
|
+
end
|
|
22
|
+
atom_count = atom_field.to_i
|
|
23
|
+
bond_count = bond_field.to_i
|
|
24
|
+
|
|
25
|
+
atoms = parse_atoms(atom_count)
|
|
26
|
+
bonds = parse_bonds(bond_count)
|
|
27
|
+
properties = parse_properties
|
|
28
|
+
|
|
29
|
+
apply_legacy_charge!(atoms)
|
|
30
|
+
apply_charges!(atoms, properties[:charges])
|
|
31
|
+
apply_isotopes!(atoms, properties[:isotopes])
|
|
32
|
+
|
|
33
|
+
adjacency = Array.new(atoms.length) { {} }
|
|
34
|
+
bonds.each do |bond|
|
|
35
|
+
adjacency[bond[:from]][bond[:to]] = bond[:kind]
|
|
36
|
+
adjacency[bond[:to]][bond[:from]] = bond[:kind]
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
# V2000 carries aromaticity on bonds (type 4); the model
|
|
40
|
+
# carries it on atoms and bonds, so atoms touching an
|
|
41
|
+
# aromatic bond are marked aromatic.
|
|
42
|
+
bonds.each do |bond|
|
|
43
|
+
next unless bond[:kind] == :aromatic
|
|
44
|
+
|
|
45
|
+
atoms[bond[:from]].aromatic = true
|
|
46
|
+
atoms[bond[:to]].aromatic = true
|
|
47
|
+
end
|
|
48
|
+
|
|
49
|
+
Model::Molecule.new(
|
|
50
|
+
nodes: Structure::Linearizer.new(atoms: atoms, edges: adjacency_to_edges(adjacency)).nodes
|
|
51
|
+
)
|
|
52
|
+
end
|
|
53
|
+
|
|
54
|
+
private
|
|
55
|
+
|
|
56
|
+
def field(line_index, start, length)
|
|
57
|
+
(@lines[line_index] || "")[start, length].to_s.strip
|
|
58
|
+
end
|
|
59
|
+
|
|
60
|
+
def parse_atoms(count)
|
|
61
|
+
(1..count).map do |i|
|
|
62
|
+
line_index = 3 + i
|
|
63
|
+
line = @lines[line_index]
|
|
64
|
+
raise ParseError, "truncated atom block (expected #{count} atoms)" if line.nil?
|
|
65
|
+
|
|
66
|
+
x = line[0, 10].to_f
|
|
67
|
+
y = line[10, 10].to_f
|
|
68
|
+
z = line[20, 10].to_f
|
|
69
|
+
element = line[31, 3].to_s.strip
|
|
70
|
+
mass_diff = line[34, 2].to_i
|
|
71
|
+
raise ParseError, "atom #{i} has no element symbol" if element.empty?
|
|
72
|
+
|
|
73
|
+
Model::Atom.new(
|
|
74
|
+
element: element,
|
|
75
|
+
x2: x, y2: y, z2: z,
|
|
76
|
+
isotope: isotope_from_mass_diff(element, mass_diff)
|
|
77
|
+
)
|
|
78
|
+
end
|
|
79
|
+
end
|
|
80
|
+
|
|
81
|
+
def parse_bonds(count)
|
|
82
|
+
(1..count).map do |i|
|
|
83
|
+
line = @lines[3 + atoms_count + i]
|
|
84
|
+
raise ParseError, "truncated bond block (expected #{count} bonds)" if line.nil?
|
|
85
|
+
|
|
86
|
+
from = line[0, 3].to_i - 1
|
|
87
|
+
to = line[3, 3].to_i - 1
|
|
88
|
+
type = line[6, 3].to_i
|
|
89
|
+
stereo = line[9, 3].to_i
|
|
90
|
+
raise ParseError, "bond #{i} has out-of-range atom indexes" if from.negative? || to.negative?
|
|
91
|
+
|
|
92
|
+
{ from: from, to: to, kind: bond_kind(type, stereo) }
|
|
93
|
+
end
|
|
94
|
+
end
|
|
95
|
+
|
|
96
|
+
def parse_properties
|
|
97
|
+
charges = {}
|
|
98
|
+
isotopes = {}
|
|
99
|
+
@lines.each do |line|
|
|
100
|
+
if line.start_with?("M CHG")
|
|
101
|
+
parts = line[6..].split
|
|
102
|
+
_count = parts[0].to_i
|
|
103
|
+
parts[1..].each_slice(2) do |idx, charge|
|
|
104
|
+
charges[idx.to_i - 1] = charge.to_i if idx && charge
|
|
105
|
+
end
|
|
106
|
+
elsif line.start_with?("M ISO")
|
|
107
|
+
parts = line[6..].split
|
|
108
|
+
parts[1..].each_slice(2) do |idx, mass|
|
|
109
|
+
isotopes[idx.to_i - 1] = mass.to_i if idx && mass
|
|
110
|
+
end
|
|
111
|
+
end
|
|
112
|
+
end
|
|
113
|
+
{ charges: charges, isotopes: isotopes }
|
|
114
|
+
end
|
|
115
|
+
|
|
116
|
+
def bond_kind(type, stereo)
|
|
117
|
+
return :wedge if stereo == 1
|
|
118
|
+
return :hash if stereo == 6
|
|
119
|
+
|
|
120
|
+
{ 1 => :single, 2 => :double, 3 => :triple, 4 => :aromatic }[type] ||
|
|
121
|
+
raise(ParseError, "unsupported molfile bond type #{type}")
|
|
122
|
+
end
|
|
123
|
+
|
|
124
|
+
# Pre-CHG charge column (0-based 36, 3): nonzero values 1..4
|
|
125
|
+
# mean +1..+4, 5..7 mean -1..-3. `M CHG` overrides.
|
|
126
|
+
def apply_legacy_charge!(atoms)
|
|
127
|
+
(1..field(3, 0, 3).to_i).each do |i|
|
|
128
|
+
code = field(3 + i, 36, 3).to_i
|
|
129
|
+
next if code.zero?
|
|
130
|
+
|
|
131
|
+
charge = code <= 4 ? code : 4 - code
|
|
132
|
+
sign = charge.negative? ? "-" : "+"
|
|
133
|
+
atoms[i - 1].charge = charge.abs == 1 ? sign : "#{charge.abs}#{sign}"
|
|
134
|
+
end
|
|
135
|
+
end
|
|
136
|
+
|
|
137
|
+
def apply_charges!(atoms, charges)
|
|
138
|
+
charges.each do |index, value|
|
|
139
|
+
sign = value.negative? ? "-" : "+"
|
|
140
|
+
atoms[index].charge = value.abs == 1 ? sign : "#{value.abs}#{sign}"
|
|
141
|
+
end
|
|
142
|
+
end
|
|
143
|
+
|
|
144
|
+
def apply_isotopes!(atoms, isotopes)
|
|
145
|
+
isotopes.each do |index, mass|
|
|
146
|
+
atoms[index].isotope = mass.to_s
|
|
147
|
+
end
|
|
148
|
+
end
|
|
149
|
+
|
|
150
|
+
# Mass difference encodes isotopes relative to the rounded
|
|
151
|
+
# average mass; mapping it unambiguously requires isotope
|
|
152
|
+
# tables, so v1 defers isotopes to the explicit `M ISO` block.
|
|
153
|
+
def isotope_from_mass_diff(_element, mass_diff)
|
|
154
|
+
nil
|
|
155
|
+
end
|
|
156
|
+
def atoms_count
|
|
157
|
+
field(3, 0, 3).to_i
|
|
158
|
+
end
|
|
159
|
+
|
|
160
|
+
def adjacency_to_edges(adjacency)
|
|
161
|
+
edges = []
|
|
162
|
+
adjacency.each_with_index do |neighbors, index|
|
|
163
|
+
neighbors.each { |to, kind| edges << Structure::Graph::Edge.new(from: index, to: to, kind: kind) if to > index }
|
|
164
|
+
end
|
|
165
|
+
edges
|
|
166
|
+
end
|
|
167
|
+
end
|
|
168
|
+
end
|
|
169
|
+
end
|
|
@@ -0,0 +1,91 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module AsciiChem
|
|
4
|
+
module Molfile
|
|
5
|
+
# Model molecule → V2000 molfile. Authored x2/y2 coordinates are
|
|
6
|
+
# used when present; otherwise a deterministic 2D layout is
|
|
7
|
+
# computed (Layout walks atoms in the same order as
|
|
8
|
+
# Structure::Graph, so positions map by index). Charges and
|
|
9
|
+
# isotopes are emitted as `M CHG` / `M ISO` property lines.
|
|
10
|
+
class Writer
|
|
11
|
+
BOND_TYPES = {
|
|
12
|
+
single: 1, double: 2, triple: 3, aromatic: 4,
|
|
13
|
+
wedge: 1, hash: 1
|
|
14
|
+
}.freeze
|
|
15
|
+
BOND_STEREO = { wedge: 1, hash: 6 }.freeze
|
|
16
|
+
|
|
17
|
+
def initialize(molecule, name: nil)
|
|
18
|
+
@molecule = molecule
|
|
19
|
+
@name = name
|
|
20
|
+
end
|
|
21
|
+
|
|
22
|
+
def write
|
|
23
|
+
atoms, edges = Structure::Graph.build(@molecule)
|
|
24
|
+
raise ParseError, "molecule has no bonds — a formula is not a structure" if edges.empty?
|
|
25
|
+
|
|
26
|
+
lines = []
|
|
27
|
+
lines << @name.to_s
|
|
28
|
+
lines << " AsciiChem"
|
|
29
|
+
lines << ""
|
|
30
|
+
lines << format("%3d%3d 0 0 0 0 0 0 0 0999 V2000", atoms.length, edges.length)
|
|
31
|
+
|
|
32
|
+
layout = nil
|
|
33
|
+
atoms.each_with_index do |atom, index|
|
|
34
|
+
x, y = coordinates(atom, index, atoms)
|
|
35
|
+
lines << format("%10.4f%10.4f%10.4f %-3s 0 0 0 0 0 0 0 0 0 0 0 0",
|
|
36
|
+
x, y, atom.z2 || 0.0, atom.element)
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
edges.each do |edge|
|
|
40
|
+
type = BOND_TYPES.fetch(edge.kind) do
|
|
41
|
+
raise ParseError, "#{edge.kind} bonds have no molfile V2000 type"
|
|
42
|
+
end
|
|
43
|
+
stereo = BOND_STEREO.fetch(edge.kind, 0)
|
|
44
|
+
lines << format("%3d%3d%3d%3d 0 0 0 0 0 0 0",
|
|
45
|
+
edge.from + 1, edge.to + 1, type, stereo)
|
|
46
|
+
end
|
|
47
|
+
|
|
48
|
+
lines << property_line("M CHG", charge_pairs(atoms))
|
|
49
|
+
lines << property_line("M ISO", isotope_pairs(atoms))
|
|
50
|
+
lines << "M END"
|
|
51
|
+
lines.compact.join("\n") << "\n"
|
|
52
|
+
end
|
|
53
|
+
|
|
54
|
+
private
|
|
55
|
+
|
|
56
|
+
# Authored coordinates win; otherwise positions from the
|
|
57
|
+
# deterministic 2D layout (same walk order as the graph).
|
|
58
|
+
def coordinates(atom, index, atoms)
|
|
59
|
+
return [atom.x2, atom.y2] if atom.x2 && atom.y2
|
|
60
|
+
|
|
61
|
+
@layout ||= AsciiChem::Layout.layout(@molecule)
|
|
62
|
+
placed = @layout.atoms[index]
|
|
63
|
+
placed ? [placed.x, placed.y] : [0.0, 0.0]
|
|
64
|
+
end
|
|
65
|
+
|
|
66
|
+
def charge_pairs(atoms)
|
|
67
|
+
atoms.each_with_index
|
|
68
|
+
.filter_map { |atom, i| [i + 1, charge_value(atom.charge)] if atom.charge }
|
|
69
|
+
end
|
|
70
|
+
|
|
71
|
+
def isotope_pairs(atoms)
|
|
72
|
+
atoms.each_with_index
|
|
73
|
+
.filter_map { |atom, i| [i + 1, atom.isotope.to_i] if atom.isotope }
|
|
74
|
+
end
|
|
75
|
+
|
|
76
|
+
# The model's number-then-sign charge ("2+") → signed integer.
|
|
77
|
+
def charge_value(charge)
|
|
78
|
+
count = charge[/\A\d+/]&.to_i || 1
|
|
79
|
+
charge.end_with?("-") ? -count : count
|
|
80
|
+
end
|
|
81
|
+
|
|
82
|
+
def property_line(prefix, pairs)
|
|
83
|
+
return nil if pairs.empty?
|
|
84
|
+
|
|
85
|
+
pairs.reduce(+"#{prefix}%3d" % pairs.length) do |line, (idx, value)|
|
|
86
|
+
line << format("%4d%4d", idx, value)
|
|
87
|
+
end
|
|
88
|
+
end
|
|
89
|
+
end
|
|
90
|
+
end
|
|
91
|
+
end
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module AsciiChem
|
|
4
|
+
# Molfile (CTfile V2000) ingestion and emission (TODO.v2 09;
|
|
5
|
+
# TODO.impl 57). Molfile is the highest-fidelity structure path:
|
|
6
|
+
# atom coordinates are preserved on the model (x2/y2/z2), charges
|
|
7
|
+
# via the CHG property block, isotopes via mass difference or the
|
|
8
|
+
# ISO block, bond stereo codes 1/6 map to wedge/hash bonds.
|
|
9
|
+
module Molfile
|
|
10
|
+
autoload :Parser, "asciichem/molfile/parser"
|
|
11
|
+
autoload :Writer, "asciichem/molfile/writer"
|
|
12
|
+
|
|
13
|
+
class << self
|
|
14
|
+
# Parses a V2000 molfile into a Model::Molecule.
|
|
15
|
+
def parse(text)
|
|
16
|
+
Parser.new(text).parse
|
|
17
|
+
end
|
|
18
|
+
|
|
19
|
+
# Emits a V2000 molfile. Uses authored x2/y2 coordinates;
|
|
20
|
+
# computes a 2D layout otherwise.
|
|
21
|
+
def write(molecule, name: nil)
|
|
22
|
+
Writer.new(molecule, name: name).write
|
|
23
|
+
end
|
|
24
|
+
end
|
|
25
|
+
end
|
|
26
|
+
end
|