asciichem 0.24.0 → 0.26.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +27 -1
- data/benchmarks/README.md +56 -0
- data/benchmarks/engines.rb +31 -0
- data/lib/asciichem/citation.rb +51 -0
- data/lib/asciichem/cli.rb +27 -0
- data/lib/asciichem/formatter/text.rb +1 -1
- data/lib/asciichem/version.rb +1 -1
- metadata +3 -1
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: 54b5e13e60ca4236628c8b8a145add5991c347747b7d7e2aa24f6acda6d89ad1
|
|
4
|
+
data.tar.gz: 7b8b2380ea9e5030065958cf35148da3addf2c6dcfb3dd1428066fee088dc410
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: 0f3f2ae54757260f1ad644d77415a557b16ccdcb15d80829775362fbd2194de3351327babf11ed662676fcfefcd6898a3e9d942928e4a831ac770bb8339846fe
|
|
7
|
+
data.tar.gz: 360208b6be6fc9666505610618e7a029711595c9ff353e3ef391e09f80619e47cafbe1304c15e2dffac7c15d436f016094e039ce6a02480393f967bbc9ab1edc
|
data/CHANGELOG.md
CHANGED
|
@@ -3,6 +3,30 @@
|
|
|
3
3
|
All notable changes to AsciiChem are documented here.
|
|
4
4
|
This project follows [Semantic Versioning](https://semver.org/).
|
|
5
5
|
|
|
6
|
+
## [0.26.0] - 2026-09-14
|
|
7
|
+
|
|
8
|
+
### Added
|
|
9
|
+
- The cite syntax (TODO.impl 45): `@cite("pubchem")` on a molecule
|
|
10
|
+
declares which source to cite it from; `Citation.for_molecule`
|
|
11
|
+
resolves the molecule's identifiers (registry keys preferred over
|
|
12
|
+
names) and emits one bibitem per cited source. Zero grammar
|
|
13
|
+
changes — `@cite` rides the property-annotation form by design.
|
|
14
|
+
|
|
15
|
+
### Fixed
|
|
16
|
+
- Molecule annotations now canonicalise with spaces between them
|
|
17
|
+
(`@name("Water") @cas("...")`), matching the TypeScript and Python
|
|
18
|
+
canonicalisers; Ruby was the outlier joining them without a
|
|
19
|
+
separator.
|
|
20
|
+
|
|
21
|
+
## [0.25.0] - 2026-09-14
|
|
22
|
+
|
|
23
|
+
### Added
|
|
24
|
+
- CLI `cite` — resolve a substance and emit the dataset-type Relaton
|
|
25
|
+
bibitem XML (the citation track's user-facing entry point).
|
|
26
|
+
- Cross-engine parsing benchmarks (`benchmarks/engines.rb`,
|
|
27
|
+
README with Ruby/TS/Python numbers) and the parsanol investigation
|
|
28
|
+
verdict (compat shim ~6x slower than parslet; not adopted).
|
|
29
|
+
|
|
6
30
|
## [0.24.0] - 2026-09-13
|
|
7
31
|
|
|
8
32
|
### Added
|
|
@@ -426,7 +450,9 @@ This project follows [Semantic Versioning](https://semver.org/).
|
|
|
426
450
|
`version`.
|
|
427
451
|
- Comprehensive RSpec suite with round-trip conformance.
|
|
428
452
|
|
|
429
|
-
[Unreleased]: https://github.com/asciichem/asciichem-ruby/compare/v0.
|
|
453
|
+
[Unreleased]: https://github.com/asciichem/asciichem-ruby/compare/v0.26.0...HEAD
|
|
454
|
+
[0.26.0]: https://github.com/asciichem/asciichem-ruby/compare/v0.25.0...v0.26.0
|
|
455
|
+
[0.25.0]: https://github.com/asciichem/asciichem-ruby/compare/v0.24.0...v0.25.0
|
|
430
456
|
[0.24.0]: https://github.com/asciichem/asciichem-ruby/compare/v0.23.0...v0.24.0
|
|
431
457
|
[0.23.0]: https://github.com/asciichem/asciichem-ruby/compare/v0.22.0...v0.23.0
|
|
432
458
|
[0.22.0]: https://github.com/asciichem/asciichem-ruby/compare/v0.21.0...v0.22.0
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
# Parsing benchmarks
|
|
2
|
+
|
|
3
|
+
Shared workload (identical inputs in every implementation) so numbers
|
|
4
|
+
are comparable across engines:
|
|
5
|
+
|
|
6
|
+
```ruby
|
|
7
|
+
WORKLOAD = ["H_2O", "Ca^2+", "SO_4^2-", "(R)-CH_3CH(OH)COOH",
|
|
8
|
+
"2H_2 + O_2 -> 2H_2O", "N_2 + 3H_2 <=>[Fe][400C] 2NH_3",
|
|
9
|
+
"C1-C-C-C-C-C1", "CH_3-CH_2-OH",
|
|
10
|
+
"^14C @name(\"carbon-14\") @cas(\"14104-86-4\")",
|
|
11
|
+
"A ->[heat] B ->[cool] C"]
|
|
12
|
+
```
|
|
13
|
+
|
|
14
|
+
| Engine | Batch (10 inputs) | Per input | Notes |
|
|
15
|
+
|---|---|---|---|
|
|
16
|
+
| Ruby (parslet), 3.4.8 arm64 | 29.1 ms | ~2.9 ms | `bundle exec ruby benchmarks/engines.rb` |
|
|
17
|
+
| Ruby + parse+text | 35.3 ms | ~3.5 ms | round-trip adds the formatter |
|
|
18
|
+
| TypeScript (peggy), Node 24 | 0.19 ms | ~19 µs | `npm run bench` (asciichem-ts) |
|
|
19
|
+
| Python (RD), 3.10 | 3.94 ms | ~394 µs | `python benchmarks/engines.py` (asciichem-py) |
|
|
20
|
+
|
|
21
|
+
The peggy engine is ~15x faster than parslet and ~20x faster than the
|
|
22
|
+
Python recursive-descent parser on this workload — the grammar-port
|
|
23
|
+
TS implementation did not trade away speed. Same machine (arm64),
|
|
24
|
+
single-run medians; treat as order-of-magnitude comparison.
|
|
25
|
+
|
|
26
|
+
## Parsanol investigation (2026-09-14)
|
|
27
|
+
|
|
28
|
+
[Parsanol](https://github.com/parsanol/parsanol-ruby) (Ribose's
|
|
29
|
+
parslet-alternative PEG library with a Rust native core) was evaluated
|
|
30
|
+
as a drop-in speedup for the reference grammar:
|
|
31
|
+
|
|
32
|
+
1. **Parslet-compat shim** (zero code change — re-parent the grammar
|
|
33
|
+
onto `Parsanol::Parslet::Parser`): the identical grammar runs
|
|
34
|
+
unchanged (10/10 workload inputs), but measures **~6x slower**
|
|
35
|
+
than parslet (12.2 s vs 2.05 s per 300x10 parses, Ruby 3.4.8,
|
|
36
|
+
arm64). The shim is a compatibility layer, not the fast path.
|
|
37
|
+
2. **Native Parsanol DSL** (subset micro-benchmark): constructs run,
|
|
38
|
+
but boundary semantics differ from parslet on greedy-regex +
|
|
39
|
+
`maybe`/`repeat` interaction (`SO_4^2-` parses under parslet,
|
|
40
|
+
fails under Parsanol native). A full port would need per-rule
|
|
41
|
+
revalidation against the whole corpus with no measured win to
|
|
42
|
+
justify it yet.
|
|
43
|
+
|
|
44
|
+
**Verdict: not adopted yet.** Two corrections to the spike (tracked
|
|
45
|
+
upstream in parsanol-ruby#25):
|
|
46
|
+
|
|
47
|
+
1. The native-DSL micro-benchmark ran in Parsanol's default `:ruby`
|
|
48
|
+
mode — the Rust core (`:native`) was never engaged, so the native
|
|
49
|
+
path is unmeasured, not disproven.
|
|
50
|
+
2. The `SO_4^2-` failure is a candidate upstream bug (`repeat` of a
|
|
51
|
+
`maybe`-prefixed sequence fails at end-of-input; minimal repro in
|
|
52
|
+
the issue).
|
|
53
|
+
|
|
54
|
+
Revisit trigger unchanged: engage the native backend for full
|
|
55
|
+
grammars, fix the repetition-termination bug, and beat parslet on
|
|
56
|
+
this workload — then re-run the corpus against the port.
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
# Cross-engine parsing benchmark (shared inputs across Ruby/TS/Python
|
|
4
|
+
# so numbers are comparable). Measures full parse (+ Text round-trip
|
|
5
|
+
# where cheap) over the canonical workload, reporting ops/sec and
|
|
6
|
+
# µs/op. Run: bundle exec ruby benchmarks/engines.rb
|
|
7
|
+
require "benchmark/ips"
|
|
8
|
+
require "asciichem"
|
|
9
|
+
|
|
10
|
+
WORKLOAD = [
|
|
11
|
+
"H_2O",
|
|
12
|
+
"Ca^2+",
|
|
13
|
+
"SO_4^2-",
|
|
14
|
+
"(R)-CH_3CH(OH)COOH",
|
|
15
|
+
"2H_2 + O_2 -> 2H_2O",
|
|
16
|
+
"N_2 + 3H_2 <=>[Fe][400C] 2NH_3",
|
|
17
|
+
"C1-C-C-C-C-C1",
|
|
18
|
+
"CH_3-CH_2-OH",
|
|
19
|
+
"^14C @name(\"carbon-14\") @cas(\"14104-86-4\")",
|
|
20
|
+
"A ->[heat] B ->[cool] C",
|
|
21
|
+
].freeze
|
|
22
|
+
|
|
23
|
+
Benchmark.ips do |x|
|
|
24
|
+
x.report("parse x10 (parslet)") do
|
|
25
|
+
WORKLOAD.each { |s| AsciiChem.parse(s) }
|
|
26
|
+
end
|
|
27
|
+
x.report("parse+text x10") do
|
|
28
|
+
WORKLOAD.each { |s| AsciiChem.parse(s).to_text }
|
|
29
|
+
end
|
|
30
|
+
x.compare!
|
|
31
|
+
end
|
data/lib/asciichem/citation.rb
CHANGED
|
@@ -81,6 +81,57 @@ module AsciiChem
|
|
|
81
81
|
bibitem(substance).to_xml
|
|
82
82
|
end
|
|
83
83
|
|
|
84
|
+
# The cite syntax (TODO.impl 45): a molecule annotated
|
|
85
|
+
# `@cite("pubchem")` (a property annotation — the grammar needs
|
|
86
|
+
# no extension) declares *which source to cite it from*. This
|
|
87
|
+
# resolves the molecule's identifiers and emits one bibitem per
|
|
88
|
+
# cited source. Returns [[source, bibitem]] pairs; empty when the
|
|
89
|
+
# molecule has no @cite annotations.
|
|
90
|
+
#
|
|
91
|
+
# AsciiChem.parse('H_2O @name("water") @cite("pubchem")')
|
|
92
|
+
# AsciiChem::Citation.for_molecule(formula.nodes.first).map(&:last)
|
|
93
|
+
def for_molecule(molecule, cache: nil, fetch: nil)
|
|
94
|
+
sources = citation_sources(molecule)
|
|
95
|
+
return [] if sources.empty?
|
|
96
|
+
|
|
97
|
+
convention, value = lookup_key(molecule)
|
|
98
|
+
unless value
|
|
99
|
+
raise Error,
|
|
100
|
+
"molecule carries no resolvable identifier for citation " \
|
|
101
|
+
"(annotate @cas/@inchikey/@smiles or @name)"
|
|
102
|
+
end
|
|
103
|
+
|
|
104
|
+
sources.filter_map do |source|
|
|
105
|
+
substance = AsciiChem::Resolver[source].new.resolve(
|
|
106
|
+
value: value, convention: convention, cache: cache, fetch: fetch)
|
|
107
|
+
next unless substance
|
|
108
|
+
|
|
109
|
+
[source, bibitem(substance)]
|
|
110
|
+
end
|
|
111
|
+
end
|
|
112
|
+
|
|
113
|
+
private
|
|
114
|
+
|
|
115
|
+
# The property annotation whose title is "cite": values are the
|
|
116
|
+
# source names to cite from.
|
|
117
|
+
def citation_sources(molecule)
|
|
118
|
+
molecule.properties
|
|
119
|
+
.select { |p| p.title == "cite" && p.value }
|
|
120
|
+
.map(&:value)
|
|
121
|
+
end
|
|
122
|
+
|
|
123
|
+
# First identifier the resolver can look up by, in preference
|
|
124
|
+
# order: unambiguous registry keys before names.
|
|
125
|
+
def lookup_key(molecule)
|
|
126
|
+
identifier = molecule.identifiers.find { |i| %w[cas inchikey pubchem-cid].include?(i.convention) }
|
|
127
|
+
return [identifier.convention, identifier.value] if identifier
|
|
128
|
+
|
|
129
|
+
name = molecule.names.first
|
|
130
|
+
return ["name", name.content] if name
|
|
131
|
+
|
|
132
|
+
nil
|
|
133
|
+
end
|
|
134
|
+
|
|
84
135
|
private
|
|
85
136
|
|
|
86
137
|
def title_base(substance)
|
data/lib/asciichem/cli.rb
CHANGED
|
@@ -119,6 +119,33 @@ module AsciiChem
|
|
|
119
119
|
exit 3
|
|
120
120
|
end
|
|
121
121
|
|
|
122
|
+
desc "cite --cas X | --name X | ...", "Resolve a substance and emit a dataset-type Relaton bibitem (XML)"
|
|
123
|
+
method_option :cas, type: :string, desc: "CAS registry number"
|
|
124
|
+
method_option :name, type: :string, desc: "Substance name"
|
|
125
|
+
method_option :cid, type: :string, desc: "PubChem CID"
|
|
126
|
+
method_option :inchikey, type: :string, desc: "InChIKey"
|
|
127
|
+
method_option :smiles, type: :string, desc: "SMILES"
|
|
128
|
+
method_option :source, type: :string, default: "pubchem", desc: "Resolver source"
|
|
129
|
+
method_option :refresh, type: :boolean, default: false, desc: "Bypass the cache"
|
|
130
|
+
def cite
|
|
131
|
+
convention, value = %i[cas name cid inchikey smiles]
|
|
132
|
+
.filter_map { |k| [k, options[k.to_s]] if options[k.to_s] }
|
|
133
|
+
.first
|
|
134
|
+
raise AsciiChem::Error, "give one of --cas/--name/--cid/--inchikey/--smiles" unless value
|
|
135
|
+
|
|
136
|
+
convention = { cas: "cas", name: "name", cid: "pubchem-cid",
|
|
137
|
+
inchikey: "inchikey", smiles: "smiles" }.fetch(convention)
|
|
138
|
+
substance = AsciiChem::Resolver[options[:source]].new.resolve(
|
|
139
|
+
value: value, convention: convention, refresh: options[:refresh]
|
|
140
|
+
)
|
|
141
|
+
raise AsciiChem::Error, "#{options[:source]} does not know #{value.inspect}" unless substance
|
|
142
|
+
|
|
143
|
+
puts AsciiChem::Citation.to_xml(substance)
|
|
144
|
+
rescue AsciiChem::Error => e
|
|
145
|
+
warn "Cite error: #{e.message}"
|
|
146
|
+
exit 4
|
|
147
|
+
end
|
|
148
|
+
|
|
122
149
|
desc "validate -i INPUT", "Offline identifier validation"
|
|
123
150
|
method_option :input, aliases: "-i", type: :string, required: true
|
|
124
151
|
def validate
|
|
@@ -38,7 +38,7 @@ module AsciiChem
|
|
|
38
38
|
molecule.labels.each { |l| parts << %(@label("#{l.value}")) if l.value }
|
|
39
39
|
molecule.properties.each { |p| parts << %(@#{p.title}("#{p.value}")) if p.title && p.value }
|
|
40
40
|
molecule.metadata.each { |m| parts << %(@meta("#{m.name}","#{m.content}")) }
|
|
41
|
-
parts.empty? ? "" : " #{parts.join}"
|
|
41
|
+
parts.empty? ? "" : " #{parts.join(" ")}"
|
|
42
42
|
end
|
|
43
43
|
|
|
44
44
|
def visit_atom(atom)
|
data/lib/asciichem/version.rb
CHANGED
metadata
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
--- !ruby/object:Gem::Specification
|
|
2
2
|
name: asciichem
|
|
3
3
|
version: !ruby/object:Gem::Version
|
|
4
|
-
version: 0.
|
|
4
|
+
version: 0.26.0
|
|
5
5
|
platform: ruby
|
|
6
6
|
authors:
|
|
7
7
|
- Ribose Inc.
|
|
@@ -172,8 +172,10 @@ files:
|
|
|
172
172
|
- RELEASING.md
|
|
173
173
|
- Rakefile
|
|
174
174
|
- asciichem.gemspec
|
|
175
|
+
- benchmarks/README.md
|
|
175
176
|
- benchmarks/RESULTS.md
|
|
176
177
|
- benchmarks/benchmark.rb
|
|
178
|
+
- benchmarks/engines.rb
|
|
177
179
|
- exe/asciichem
|
|
178
180
|
- lib/asciichem.rb
|
|
179
181
|
- lib/asciichem/citation.rb
|