asciichem 0.29.1 → 0.29.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.github/workflows/release.yml +9 -0
- data/CHANGELOG.md +22 -0
- data/Gemfile +1 -0
- data/asciichem.gemspec +6 -4
- data/benchmarks/README.md +117 -6
- data/lib/asciichem/citation.rb +127 -43
- data/lib/asciichem/version.rb +1 -1
- metadata +11 -23
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: 6aed1bf1e1bf05bd4f3459f3bf147d39f08f283dc49b8452040643171e836a30
|
|
4
|
+
data.tar.gz: 557b7506d58e8505c6b9e3ffa36da6ef7025e054850eba7a9fdee7e03503c337
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: b3ff4937b6f92f253c16f03d414ef4fa177ab0df85fea38c61be07e83a04e7889c2485e4b4d8b7dd3c2a7e0042711e533cdf5968e11abc130d68d21620e6ed07
|
|
7
|
+
data.tar.gz: dfeada783a748286cf51fbbb9cbaa0a8f7a485af352b511dd54a67d53fce0ec77c9b58d339d980853664ff58438f0e5be56ae6632802d70a44e12967d5418285
|
|
@@ -34,6 +34,15 @@ jobs:
|
|
|
34
34
|
with:
|
|
35
35
|
ruby-version: "3.4"
|
|
36
36
|
bundler-cache: true
|
|
37
|
+
# CI never pushes to git (read-only contents). `rake release`
|
|
38
|
+
# attempts `git push origin main` after publishing unless the
|
|
39
|
+
# version tag already exists locally — bundler then prints
|
|
40
|
+
# "Tag vX has already been created" and skips its git stage
|
|
41
|
+
# entirely (this is how 0.29.0/0.29.1 released). Pre-create the
|
|
42
|
+
# tag so the gem push is the only remote operation; tags on the
|
|
43
|
+
# remote remain the maintainer's.
|
|
44
|
+
- name: Pre-create the release tag (skips rake's git stage)
|
|
45
|
+
run: git tag "v${{ inputs.version }}"
|
|
37
46
|
# Builds and pushes using the GitHub OIDC identity — no API keys.
|
|
38
47
|
- uses: rubygems/release-gem@v1
|
|
39
48
|
- name: Summary
|
data/CHANGELOG.md
CHANGED
|
@@ -3,6 +3,28 @@
|
|
|
3
3
|
All notable changes to AsciiChem are documented here.
|
|
4
4
|
This project follows [Semantic Versioning](https://semver.org/).
|
|
5
5
|
|
|
6
|
+
## [0.29.3] - 2026-09-22
|
|
7
|
+
|
|
8
|
+
### Changed
|
|
9
|
+
- Dependency floors raised to the corpus-validated versions,
|
|
10
|
+
expressed pessimistically (`~>`): `lutaml-model ~> 0.8` (from
|
|
11
|
+
`>= 0.8, < 2`), `relaton-bib ~> 2.1` (from `>= 0.1, < 3`),
|
|
12
|
+
`plurimath ~> 0.11`, `nokogiri ~> 1.18`; `chemicalml`, `elkrb`,
|
|
13
|
+
`mml`, `parslet`, `thor` unchanged. Full suite green under the
|
|
14
|
+
raised floors (1985 examples).
|
|
15
|
+
|
|
16
|
+
## [0.29.2] - 2026-09-17
|
|
17
|
+
|
|
18
|
+
### Changed
|
|
19
|
+
- `relaton-bib` constraint widened to `>= 0.1, < 3`: asciichem now
|
|
20
|
+
co-resolves with current metanorma gems (metanorma-standoc and
|
|
21
|
+
friends require relaton-bib 2). The citation track speaks both
|
|
22
|
+
major lines through a single `Citation::RelatonApi` seam —
|
|
23
|
+
relaton-bib 1 (`RelatonBib`) and relaton-bib 2
|
|
24
|
+
(`Relaton::Bib` typed models) both build and serialize the
|
|
25
|
+
dataset-type bibitems; profile data is version-independent.
|
|
26
|
+
- nil resolver links no longer emit an empty `<uri>` element.
|
|
27
|
+
|
|
6
28
|
## [0.29.1] - 2026-09-16
|
|
7
29
|
|
|
8
30
|
### Added
|
data/Gemfile
CHANGED
data/asciichem.gemspec
CHANGED
|
@@ -33,14 +33,16 @@ Gem::Specification.new do |spec|
|
|
|
33
33
|
spec.executables = spec.files.grep(%r{^exe/}) { |f| File.basename(f) }
|
|
34
34
|
spec.require_paths = ["lib"]
|
|
35
35
|
|
|
36
|
+
# Floors are the versions the corpus suite validates against
|
|
37
|
+
# (pessimistic ~>); raise them only with a full-suite run.
|
|
36
38
|
spec.add_dependency "chemicalml", "~> 0.3.0"
|
|
37
39
|
spec.add_dependency "elkrb", "~> 1.0"
|
|
38
|
-
spec.add_dependency "lutaml-model", "
|
|
40
|
+
spec.add_dependency "lutaml-model", "~> 0.8"
|
|
39
41
|
spec.add_dependency "mml", "~> 2.3"
|
|
40
|
-
spec.add_dependency "nokogiri", "~> 1.
|
|
42
|
+
spec.add_dependency "nokogiri", "~> 1.18"
|
|
41
43
|
spec.add_dependency "parslet", "~> 2.0"
|
|
42
|
-
spec.add_dependency "relaton-bib", "
|
|
43
|
-
spec.add_dependency "plurimath", "~> 0.
|
|
44
|
+
spec.add_dependency "relaton-bib", "~> 2.1"
|
|
45
|
+
spec.add_dependency "plurimath", "~> 0.11"
|
|
44
46
|
spec.add_dependency "thor", "~> 1.3"
|
|
45
47
|
|
|
46
48
|
spec.add_development_dependency "json_schemer", "~> 2.4"
|
data/benchmarks/README.md
CHANGED
|
@@ -113,9 +113,120 @@ unaffected; `@next_id` remains unfixed upstream but no longer fires
|
|
|
113
113
|
on corpus inputs. The re-check-3 verdict below is superseded — the
|
|
114
114
|
engine IS switchable and shipped (TODO.impl 64).
|
|
115
115
|
|
|
116
|
-
**Verdict:
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
116
|
+
**Verdict: superseded — the engine IS switchable and shipped**
|
|
117
|
+
(asciichem 0.29.0, TODO.impl 64).
|
|
118
|
+
|
|
119
|
+
### Re-check 6 (2026-09-16, parsanol 1.3.18)
|
|
120
|
+
|
|
121
|
+
Gate still **221/221** through the shipped engine, but the "one
|
|
122
|
+
decode path" rework **regressed compat-layer throughput ~60%** for
|
|
123
|
+
this grammar: 7.2 ms/batch (138 i/s) vs 4.3-4.6 ms on 1.3.15/16,
|
|
124
|
+
with the parslet control stable across sessions (11-14 ms
|
|
125
|
+
throughout). The `H2`/`_2O` acceptance divergence also persists.
|
|
126
|
+
Reported upstream (parsanol-ruby#25, fourth comment). We stay on
|
|
127
|
+
the shipped engine; users pinning parsanol for speed should prefer
|
|
128
|
+
1.3.16/1.3.17 until the regression is addressed.
|
|
129
|
+
|
|
130
|
+
### Re-check 7 (2026-09-16, parsanol 1.3.20)
|
|
131
|
+
|
|
132
|
+
Three upstream issues closed since 1.3.18:
|
|
133
|
+
|
|
134
|
+
- **#38** (`Dynamic.register` `@next_id` collision panicking the
|
|
135
|
+
Rust core) — fixed; no panic during full-corpus run.
|
|
136
|
+
- **#37** ("one decode path" throughput regression) — fixed as a
|
|
137
|
+
side effect of the optimizer acceptance fix in #39; throughput on
|
|
138
|
+
this grammar is back to and ahead of 1.3.15/16 levels.
|
|
139
|
+
- **#39** (optimizer Str/Re run-merging changed sequence-boundary
|
|
140
|
+
acceptance) — root-caused to Re-run regex-source concatenation
|
|
141
|
+
(proven unsafe: `"a|"+"b"` → `"a|b"` accepts `"a"`); Re runs now
|
|
142
|
+
stay unmerged, Str-run merging stays. Spec-level decision
|
|
143
|
+
recorded: the optimizer may never alter acceptance.
|
|
144
|
+
|
|
145
|
+
Validation against 1.3.20:
|
|
146
|
+
|
|
147
|
+
- Gate **221/221** through the shipped `ParsanolEngine` (fork-per-case,
|
|
148
|
+
no Rust aborts).
|
|
149
|
+
- Head-to-head vs parslet, same Ruby process (3 runs, ±3-15%):
|
|
150
|
+
parsanol **2.6x faster** (4.65–5.19 ms/batch vs 12.18–13.65 ms for
|
|
151
|
+
parslet). Up from the 1.7x under 1.3.18 — the #37 regression is
|
|
152
|
+
gone.
|
|
153
|
+
- Direct `H2` / `_2O` / `Ca2+` / `H22` / `O2` probe across both
|
|
154
|
+
parslet and parsanol (native and ruby backends) shows **identical
|
|
155
|
+
parse outcomes**. The earlier "divergence" framing in re-checks
|
|
156
|
+
3-6 was a misreading: AsciiChem's `hydrogen_atom` grammar rule
|
|
157
|
+
intentionally permits bare-digit subscripts after `H` ("lets users
|
|
158
|
+
write `H2O` instead of `H_2O`" — grammar_rules.rb:228-231) and
|
|
159
|
+
`isotope_marker` accepts both `^digits` and `_digits`, so `_2O`
|
|
160
|
+
parses as the isotope of `O` and round-trips as `^2O`. The
|
|
161
|
+
parsanol optimizer bug in #39 was real and is fixed, but the
|
|
162
|
+
AsciiChem repro was a misleading example — both engines agree on
|
|
163
|
+
these inputs because they share the same grammar rules.
|
|
164
|
+
|
|
165
|
+
**Verdict: shipped engine fully validated.** 2.6x speedup, 100%
|
|
166
|
+
corpus gate, all four reported upstream issues now resolved or
|
|
167
|
+
non-blocking (#36 bare repeated sibling captures remains open but
|
|
168
|
+
is worked around in `ParsanolEngine` via single `.as(...)` capture
|
|
169
|
+
wrapping).
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
### Re-check 8 (2026-09-17, parsanol 1.3.27)
|
|
173
|
+
|
|
174
|
+
All four upstream issues we filed are now closed (#36-#39; seven
|
|
175
|
+
releases since 1.3.20). Validation:
|
|
176
|
+
|
|
177
|
+
- Gate **221/221** through the shipped `ParsanolEngine`.
|
|
178
|
+
- **#36 verified fixed at the source**: the bare-repeated-sibling
|
|
179
|
+
repro (`A->B->C`) now returns parslet's array-of-segment-hashes —
|
|
180
|
+
every match preserved. Our `split_merged_formula` seam in
|
|
181
|
+
`ParsanolEngine` is therefore a compatibility no-op on current
|
|
182
|
+
parsanol (it still normalizes the merged-hash shape for older
|
|
183
|
+
parsanol lines, which the opt-in floor allows).
|
|
184
|
+
- **Perf: ratio-only this time.** The machine ran at load ~45
|
|
185
|
+
during measurement (parallel spec suites in other sessions), so
|
|
186
|
+
absolute numbers are meaningless — the parslet control itself
|
|
187
|
+
measured 15-20x slower than its quiet-machine baseline.
|
|
188
|
+
Same-process ratio: parsanol **~2.1x parslet** (8.3 vs 3.8 i/s,
|
|
189
|
+
and 8.8 vs 4.5 on the repeat), consistent with the 2.6x
|
|
190
|
+
quiet-machine figure from re-check 7.
|
|
191
|
+
|
|
192
|
+
### Re-check 9 (2026-09-22, parsanol 1.3.49)
|
|
193
|
+
|
|
194
|
+
Twenty-one releases since re-check 8, all perf-focused upstream
|
|
195
|
+
(#59 roadmap: first-set BYTE_DISPATCH 1.3.29, VM memoisation
|
|
196
|
+
1.3.35, VM phase-2 wiring 1.3.33, native dynamic bridge fixes
|
|
197
|
+
1.3.40-1.3.41, `Parsanol::IncrementalSession` 1.3.42). Validation:
|
|
198
|
+
|
|
199
|
+
- Gate **221/221** through the shipped `ParsanolEngine` (run on
|
|
200
|
+
1.3.48/1.3.49 within the same day — the line is moving fast).
|
|
201
|
+
- **Perf: ratio-only again.** Load was 34-77 during measurement
|
|
202
|
+
(parslet control itself ran 8-12 i/s vs its quiet ~75), so
|
|
203
|
+
absolute numbers are excluded. Same-process ratio across three
|
|
204
|
+
runs: parsanol **1.7-3.2x parslet** (37.7/11.7, 19.0/11.1,
|
|
205
|
+
25.5/8.4), centring ~2.5-3x — consistent with the quiet-machine
|
|
206
|
+
2.6x from re-check 7; under contention the native parse path
|
|
207
|
+
degrades less than pure-Ruby parslet.
|
|
208
|
+
- Upstream's incremental (`Parsanol::IncrementalSession`) and VM
|
|
209
|
+
memoisation work benefits the compat layer automatically; no
|
|
210
|
+
asciichem-side change needed or made.
|
|
211
|
+
|
|
212
|
+
## Leptris note (2026-09-22, 1.9.221)
|
|
213
|
+
|
|
214
|
+
Leptris is moxml's PREFERRED_ADAPTER: when installed above its
|
|
215
|
+
binding floor, lutaml-model's XML layer (our CML wire path) runs on
|
|
216
|
+
it. The version is fully transitive — lutaml-model constrains
|
|
217
|
+
`~> 1.9.178`; asciichem pins nothing. The line moves fast
|
|
218
|
+
(1.9.178 floor -> 1.9.222 within days).
|
|
219
|
+
|
|
220
|
+
- **Compatibility:** full suite **1985/0** at 1.9.221.1, including
|
|
221
|
+
every CML round-trip and three-way wire spec. 1.9.222's namespace
|
|
222
|
+
fix (`xml:space` in the interleaved lane, reported upstream by
|
|
223
|
+
Canon) does not affect our documents; no action.
|
|
224
|
+
- **Perf on our CML workload:** leptris is ~15-25% behind nokogiri
|
|
225
|
+
(round-trip 12.7 vs 15.0 i/s; emit 30.6 vs 39.8 i/s; load-noisy
|
|
226
|
+
±20%). The workload is dominated by lutaml-model's Ruby-side
|
|
227
|
+
model building, not the adapter — leptris's speed gains target
|
|
228
|
+
its native parse lanes (HTML/XQuery per its release notes).
|
|
229
|
+
Measurement caveat: forcing an adapter for A/B runs requires
|
|
230
|
+
stubbing `leptris_preferred_available?` — lutaml's
|
|
231
|
+
`detect_xml_adapter` calls `runtime_default_adapter`, which
|
|
232
|
+
ignores `default_adapter=`.
|
data/lib/asciichem/citation.rb
CHANGED
|
@@ -1,6 +1,14 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
# relaton-bib 2 renamed the entry file (relaton_bib -> relaton/bib)
|
|
4
|
+
# and reworked the namespace (RelatonBib -> Relaton::Bib). The
|
|
5
|
+
# gemspec admits both major lines, so load whichever is resolved and
|
|
6
|
+
# speak to it through RelatonApi below.
|
|
7
|
+
begin
|
|
8
|
+
require 'relaton/bib'
|
|
9
|
+
rescue LoadError
|
|
10
|
+
require 'relaton_bib'
|
|
11
|
+
end
|
|
4
12
|
|
|
5
13
|
module AsciiChem
|
|
6
14
|
# Citation track (TODO.v2 08; TODO.impl 44): a bibitem is a function
|
|
@@ -16,25 +24,25 @@ module AsciiChem
|
|
|
16
24
|
Profile = Struct.new(:publisher, :link_for, :identifier_for, keyword_init: true)
|
|
17
25
|
|
|
18
26
|
PROFILES = {
|
|
19
|
-
|
|
20
|
-
publisher:
|
|
21
|
-
link_for:
|
|
22
|
-
cid = substance.identifier_value(
|
|
27
|
+
'pubchem' => Profile.new(
|
|
28
|
+
publisher: 'PubChem, U.S. National Library of Medicine',
|
|
29
|
+
link_for: lambda do |substance|
|
|
30
|
+
cid = substance.identifier_value('pubchem-cid')
|
|
23
31
|
"https://pubchem.ncbi.nlm.nih.gov/compound/#{cid}" if cid
|
|
24
32
|
end,
|
|
25
|
-
identifier_for:
|
|
26
|
-
cid = substance.identifier_value(
|
|
33
|
+
identifier_for: lambda do |substance|
|
|
34
|
+
cid = substance.identifier_value('pubchem-cid')
|
|
27
35
|
"PubChem CID #{cid}" if cid
|
|
28
36
|
end
|
|
29
37
|
),
|
|
30
|
-
|
|
31
|
-
publisher:
|
|
32
|
-
link_for:
|
|
33
|
-
cas = substance.identifier_value(
|
|
38
|
+
'common_chemistry' => Profile.new(
|
|
39
|
+
publisher: 'CAS Common Chemistry',
|
|
40
|
+
link_for: lambda do |substance|
|
|
41
|
+
cas = substance.identifier_value('cas')
|
|
34
42
|
"https://commonchemistry.cas.org/detail?cas_rn=#{cas}" if cas
|
|
35
43
|
end,
|
|
36
|
-
identifier_for:
|
|
37
|
-
cas = substance.identifier_value(
|
|
44
|
+
identifier_for: lambda do |substance|
|
|
45
|
+
cas = substance.identifier_value('cas')
|
|
38
46
|
"CAS RN #{cas}" if cas
|
|
39
47
|
end
|
|
40
48
|
)
|
|
@@ -42,8 +50,8 @@ module AsciiChem
|
|
|
42
50
|
|
|
43
51
|
DEFAULT_PROFILE = Profile.new(
|
|
44
52
|
publisher: nil,
|
|
45
|
-
link_for: ->(_substance) {
|
|
46
|
-
identifier_for:
|
|
53
|
+
link_for: ->(_substance) {},
|
|
54
|
+
identifier_for: lambda do |substance|
|
|
47
55
|
key = substance.identifiers.first
|
|
48
56
|
"#{key.convention}: #{key.value}" if key
|
|
49
57
|
end
|
|
@@ -56,34 +64,20 @@ module AsciiChem
|
|
|
56
64
|
# carries no provenance (hand-built, not resolved).
|
|
57
65
|
def bibitem(substance)
|
|
58
66
|
provenance = substance.provenance
|
|
59
|
-
unless provenance&.source
|
|
60
|
-
raise Error, "substance has no provenance - resolve it first (AsciiChem::Resolver)"
|
|
61
|
-
end
|
|
67
|
+
raise Error, 'substance has no provenance - resolve it first (AsciiChem::Resolver)' unless provenance&.source
|
|
62
68
|
|
|
63
69
|
profile = PROFILES.fetch(provenance.source, DEFAULT_PROFILE)
|
|
64
|
-
|
|
65
|
-
type: "dataset",
|
|
66
|
-
title: [{ type: "main",
|
|
67
|
-
content: "#{title_base(substance)} - #{profile.publisher || provenance.source} substance record" }],
|
|
68
|
-
docid: [RelatonBib::DocumentIdentifier.new(
|
|
69
|
-
id: profile.identifier_for.call(substance) || "#{provenance.source} substance",
|
|
70
|
-
type: provenance.source)],
|
|
71
|
-
contributor: [{ entity: RelatonBib::Organization.new(name: profile.publisher || provenance.source),
|
|
72
|
-
role: [{ type: "publisher" }] }],
|
|
73
|
-
date: [{ type: "accessed", on: accessed_on(provenance) }],
|
|
74
|
-
link: [{ type: "src", content: profile.link_for.call(substance) }].compact,
|
|
75
|
-
keyword: substance.identifiers.map { |i| "#{i.convention}=#{i.value}" }
|
|
76
|
-
)
|
|
70
|
+
RelatonApi.dataset_bibitem(fields(substance, profile, provenance))
|
|
77
71
|
end
|
|
78
72
|
|
|
79
73
|
# Convenience: bibitem XML (what a document pipeline embeds).
|
|
80
74
|
def to_xml(substance)
|
|
81
|
-
bibitem(substance)
|
|
75
|
+
RelatonApi.to_xml(bibitem(substance))
|
|
82
76
|
end
|
|
83
77
|
|
|
84
78
|
# The cite syntax (TODO.impl 45): a molecule annotated
|
|
85
79
|
# `@cite("pubchem")` (a property annotation — the grammar needs
|
|
86
|
-
# no extension) declares *which source to cite it from
|
|
80
|
+
# no extension) declares *which source to cite it from. This
|
|
87
81
|
# resolves the molecule's identifiers and emits one bibitem per
|
|
88
82
|
# cited source. Returns [[source, bibitem]] pairs; empty when the
|
|
89
83
|
# molecule has no @cite annotations.
|
|
@@ -97,13 +91,14 @@ module AsciiChem
|
|
|
97
91
|
convention, value = lookup_key(molecule)
|
|
98
92
|
unless value
|
|
99
93
|
raise Error,
|
|
100
|
-
|
|
101
|
-
|
|
94
|
+
'molecule carries no resolvable identifier for citation ' \
|
|
95
|
+
'(annotate @cas/@inchikey/@smiles or @name)'
|
|
102
96
|
end
|
|
103
97
|
|
|
104
98
|
sources.filter_map do |source|
|
|
105
99
|
substance = AsciiChem::Resolver[source].new.resolve(
|
|
106
|
-
value: value, convention: convention, cache: cache, fetch: fetch
|
|
100
|
+
value: value, convention: convention, cache: cache, fetch: fetch
|
|
101
|
+
)
|
|
107
102
|
next unless substance
|
|
108
103
|
|
|
109
104
|
[source, bibitem(substance)]
|
|
@@ -112,11 +107,27 @@ module AsciiChem
|
|
|
112
107
|
|
|
113
108
|
private
|
|
114
109
|
|
|
110
|
+
# The version-independent field payload: one hash describing the
|
|
111
|
+
# citation, translated to Relaton objects by RelatonApi.
|
|
112
|
+
def fields(substance, profile, provenance)
|
|
113
|
+
publisher = profile.publisher || provenance.source
|
|
114
|
+
{
|
|
115
|
+
type: 'dataset',
|
|
116
|
+
title: "#{title_base(substance)} - #{publisher} substance record",
|
|
117
|
+
docid: { id: profile.identifier_for.call(substance) || "#{provenance.source} substance",
|
|
118
|
+
type: provenance.source },
|
|
119
|
+
publisher: publisher,
|
|
120
|
+
accessed_on: accessed_on(provenance),
|
|
121
|
+
link: profile.link_for.call(substance),
|
|
122
|
+
keywords: substance.identifiers.map { |i| "#{i.convention}=#{i.value}" }
|
|
123
|
+
}
|
|
124
|
+
end
|
|
125
|
+
|
|
115
126
|
# The property annotation whose title is "cite": values are the
|
|
116
127
|
# source names to cite from.
|
|
117
128
|
def citation_sources(molecule)
|
|
118
129
|
molecule.properties
|
|
119
|
-
.select { |p| p.title ==
|
|
130
|
+
.select { |p| p.title == 'cite' && p.value }
|
|
120
131
|
.map(&:value)
|
|
121
132
|
end
|
|
122
133
|
|
|
@@ -127,23 +138,96 @@ module AsciiChem
|
|
|
127
138
|
return [identifier.convention, identifier.value] if identifier
|
|
128
139
|
|
|
129
140
|
name = molecule.names.first
|
|
130
|
-
return [
|
|
141
|
+
return ['name', name.content] if name
|
|
131
142
|
|
|
132
143
|
nil
|
|
133
144
|
end
|
|
134
145
|
|
|
135
|
-
private
|
|
136
|
-
|
|
137
146
|
def title_base(substance)
|
|
138
|
-
substance.preferred_name || substance.identifier_value(
|
|
139
|
-
substance.identifier_value(
|
|
147
|
+
substance.preferred_name || substance.identifier_value('cas') ||
|
|
148
|
+
substance.identifier_value('inchikey') || 'Substance'
|
|
140
149
|
end
|
|
141
150
|
|
|
142
151
|
def accessed_on(provenance)
|
|
143
152
|
return provenance.retrieved_at[0, 10] if provenance.retrieved_at
|
|
144
153
|
|
|
145
|
-
Time.now.utc.strftime(
|
|
154
|
+
Time.now.utc.strftime('%Y-%m-%d')
|
|
155
|
+
end
|
|
156
|
+
end
|
|
157
|
+
|
|
158
|
+
# The relaton-bib version seam. Both major lines accept the same
|
|
159
|
+
# field hash (see Citation#fields) and serialize through their own
|
|
160
|
+
# API; the rest of the citation track stays version-agnostic.
|
|
161
|
+
# Adding a future major = one more module here (OCP).
|
|
162
|
+
module RelatonApi
|
|
163
|
+
module_function
|
|
164
|
+
|
|
165
|
+
def dataset_bibitem(fields)
|
|
166
|
+
(defined?(::Relaton::Bib) ? V2 : V1).build(fields)
|
|
167
|
+
end
|
|
168
|
+
|
|
169
|
+
def to_xml(item)
|
|
170
|
+
item.to_xml
|
|
171
|
+
end
|
|
172
|
+
|
|
173
|
+
# relaton-bib 1: RelatonBib::* with hash-argument constructors.
|
|
174
|
+
module V1
|
|
175
|
+
module_function
|
|
176
|
+
|
|
177
|
+
def build(fields)
|
|
178
|
+
RelatonBib::BibliographicItem.new(
|
|
179
|
+
type: fields[:type],
|
|
180
|
+
title: [{ type: 'main', content: fields[:title] }],
|
|
181
|
+
docid: [RelatonBib::DocumentIdentifier.new(id: fields[:docid][:id],
|
|
182
|
+
type: fields[:docid][:type])],
|
|
183
|
+
contributor: [{ entity: RelatonBib::Organization.new(name: fields[:publisher]),
|
|
184
|
+
role: [{ type: 'publisher' }] }],
|
|
185
|
+
date: [{ type: 'accessed', on: fields[:accessed_on] }],
|
|
186
|
+
link: fields[:link] ? [{ type: 'src', content: fields[:link] }] : [],
|
|
187
|
+
keyword: fields[:keywords]
|
|
188
|
+
)
|
|
189
|
+
end
|
|
190
|
+
end
|
|
191
|
+
|
|
192
|
+
# relaton-bib 2: Relaton::Bib::* typed models (lutaml-model).
|
|
193
|
+
# Date's XML <on> element maps to the Ruby `at` attribute;
|
|
194
|
+
# keywords carry their text in a nested vocab LocalizedString;
|
|
195
|
+
# links are source Uri entries serializing to <uri type="src">.
|
|
196
|
+
module V2
|
|
197
|
+
module_function
|
|
198
|
+
|
|
199
|
+
def build(fields)
|
|
200
|
+
Relaton::Bib::ItemData.new(
|
|
201
|
+
type: fields[:type],
|
|
202
|
+
title: [Relaton::Bib::Title.new(type: 'main', content: fields[:title])],
|
|
203
|
+
docidentifier: [docidentifier(fields[:docid])],
|
|
204
|
+
contributor: [contributor(fields[:publisher])],
|
|
205
|
+
date: [Relaton::Bib::Date.new(type: 'accessed', at: fields[:accessed_on])],
|
|
206
|
+
source: fields[:link] ? [Relaton::Bib::Uri.new(type: 'src', content: fields[:link])] : [],
|
|
207
|
+
keyword: fields[:keywords].map { |text| keyword(text) }
|
|
208
|
+
)
|
|
209
|
+
end
|
|
210
|
+
|
|
211
|
+
def docidentifier(docid)
|
|
212
|
+
Relaton::Bib::Docidentifier.new(type: docid[:type], content: docid[:id])
|
|
213
|
+
end
|
|
214
|
+
|
|
215
|
+
def contributor(publisher)
|
|
216
|
+
Relaton::Bib::Contributor.new(
|
|
217
|
+
organization: Relaton::Bib::Organization.new(
|
|
218
|
+
name: [Relaton::Bib::TypedLocalizedString.new(content: publisher)]
|
|
219
|
+
),
|
|
220
|
+
role: [Relaton::Bib::Contributor::Role.new(type: 'publisher')]
|
|
221
|
+
)
|
|
222
|
+
end
|
|
223
|
+
|
|
224
|
+
def keyword(text)
|
|
225
|
+
Relaton::Bib::Keyword.new(
|
|
226
|
+
vocab: Relaton::Bib::LocalizedString.new(content: text)
|
|
227
|
+
)
|
|
228
|
+
end
|
|
146
229
|
end
|
|
147
230
|
end
|
|
231
|
+
private_constant :RelatonApi
|
|
148
232
|
end
|
|
149
233
|
end
|
data/lib/asciichem/version.rb
CHANGED
metadata
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
--- !ruby/object:Gem::Specification
|
|
2
2
|
name: asciichem
|
|
3
3
|
version: !ruby/object:Gem::Version
|
|
4
|
-
version: 0.29.
|
|
4
|
+
version: 0.29.3
|
|
5
5
|
platform: ruby
|
|
6
6
|
authors:
|
|
7
7
|
- Ribose Inc.
|
|
@@ -41,22 +41,16 @@ dependencies:
|
|
|
41
41
|
name: lutaml-model
|
|
42
42
|
requirement: !ruby/object:Gem::Requirement
|
|
43
43
|
requirements:
|
|
44
|
-
- - "
|
|
44
|
+
- - "~>"
|
|
45
45
|
- !ruby/object:Gem::Version
|
|
46
46
|
version: '0.8'
|
|
47
|
-
- - "<"
|
|
48
|
-
- !ruby/object:Gem::Version
|
|
49
|
-
version: '2'
|
|
50
47
|
type: :runtime
|
|
51
48
|
prerelease: false
|
|
52
49
|
version_requirements: !ruby/object:Gem::Requirement
|
|
53
50
|
requirements:
|
|
54
|
-
- - "
|
|
51
|
+
- - "~>"
|
|
55
52
|
- !ruby/object:Gem::Version
|
|
56
53
|
version: '0.8'
|
|
57
|
-
- - "<"
|
|
58
|
-
- !ruby/object:Gem::Version
|
|
59
|
-
version: '2'
|
|
60
54
|
- !ruby/object:Gem::Dependency
|
|
61
55
|
name: mml
|
|
62
56
|
requirement: !ruby/object:Gem::Requirement
|
|
@@ -77,14 +71,14 @@ dependencies:
|
|
|
77
71
|
requirements:
|
|
78
72
|
- - "~>"
|
|
79
73
|
- !ruby/object:Gem::Version
|
|
80
|
-
version: '1.
|
|
74
|
+
version: '1.18'
|
|
81
75
|
type: :runtime
|
|
82
76
|
prerelease: false
|
|
83
77
|
version_requirements: !ruby/object:Gem::Requirement
|
|
84
78
|
requirements:
|
|
85
79
|
- - "~>"
|
|
86
80
|
- !ruby/object:Gem::Version
|
|
87
|
-
version: '1.
|
|
81
|
+
version: '1.18'
|
|
88
82
|
- !ruby/object:Gem::Dependency
|
|
89
83
|
name: parslet
|
|
90
84
|
requirement: !ruby/object:Gem::Requirement
|
|
@@ -103,36 +97,30 @@ dependencies:
|
|
|
103
97
|
name: relaton-bib
|
|
104
98
|
requirement: !ruby/object:Gem::Requirement
|
|
105
99
|
requirements:
|
|
106
|
-
- - "
|
|
107
|
-
- !ruby/object:Gem::Version
|
|
108
|
-
version: '0.1'
|
|
109
|
-
- - "<"
|
|
100
|
+
- - "~>"
|
|
110
101
|
- !ruby/object:Gem::Version
|
|
111
|
-
version: '2'
|
|
102
|
+
version: '2.1'
|
|
112
103
|
type: :runtime
|
|
113
104
|
prerelease: false
|
|
114
105
|
version_requirements: !ruby/object:Gem::Requirement
|
|
115
106
|
requirements:
|
|
116
|
-
- - "
|
|
117
|
-
- !ruby/object:Gem::Version
|
|
118
|
-
version: '0.1'
|
|
119
|
-
- - "<"
|
|
107
|
+
- - "~>"
|
|
120
108
|
- !ruby/object:Gem::Version
|
|
121
|
-
version: '2'
|
|
109
|
+
version: '2.1'
|
|
122
110
|
- !ruby/object:Gem::Dependency
|
|
123
111
|
name: plurimath
|
|
124
112
|
requirement: !ruby/object:Gem::Requirement
|
|
125
113
|
requirements:
|
|
126
114
|
- - "~>"
|
|
127
115
|
- !ruby/object:Gem::Version
|
|
128
|
-
version: '0.
|
|
116
|
+
version: '0.11'
|
|
129
117
|
type: :runtime
|
|
130
118
|
prerelease: false
|
|
131
119
|
version_requirements: !ruby/object:Gem::Requirement
|
|
132
120
|
requirements:
|
|
133
121
|
- - "~>"
|
|
134
122
|
- !ruby/object:Gem::Version
|
|
135
|
-
version: '0.
|
|
123
|
+
version: '0.11'
|
|
136
124
|
- !ruby/object:Gem::Dependency
|
|
137
125
|
name: thor
|
|
138
126
|
requirement: !ruby/object:Gem::Requirement
|