relaton-cli 3.0.0.pre.alpha.1 → 3.0.0.pre.alpha.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.gitignore +4 -0
- data/CLAUDE.md +403 -0
- data/Gemfile +5 -0
- data/Rakefile +15 -0
- data/docs/README.adoc +177 -1
- data/frontend/dist/404.html +18 -0
- data/frontend/dist/app.iife.js +1 -0
- data/frontend/dist/style.css +3 -0
- data/lib/relaton/cli/command.rb +116 -0
- data/lib/relaton/cli/frontend_assets.rb +58 -0
- data/lib/relaton/cli/index_item_normalizer.rb +380 -0
- data/lib/relaton/cli/index_site_generator.rb +860 -0
- data/lib/relaton/cli/subcommand_collection.rb +3 -0
- data/lib/relaton-cli.rb +5 -0
- data/relaton-cli.gemspec +13 -3
- data/templates/index/page.liquid +20 -0
- metadata +27 -9
data/lib/relaton/cli/command.rb
CHANGED
|
@@ -1,8 +1,12 @@
|
|
|
1
1
|
require "date"
|
|
2
|
+
require "tmpdir"
|
|
3
|
+
require "shellwords"
|
|
4
|
+
require "pubid"
|
|
2
5
|
require "relaton/cli/relaton_file"
|
|
3
6
|
require "relaton/cli/xml_convertor"
|
|
4
7
|
require "relaton/cli/yaml_convertor"
|
|
5
8
|
require "relaton/cli/data_fetcher"
|
|
9
|
+
require "relaton/cli/index_site_generator"
|
|
6
10
|
require "relaton/cli/subcommand_collection"
|
|
7
11
|
require "relaton/cli/subcommand_db"
|
|
8
12
|
require "fcntl"
|
|
@@ -14,6 +18,16 @@ module Relaton
|
|
|
14
18
|
class_before :relaton_config
|
|
15
19
|
class_option :verbose, aliases: :v, type: :boolean, desc: "Output warnings"
|
|
16
20
|
|
|
21
|
+
# Exit non-zero when Thor itself rejects an invocation (unknown argument,
|
|
22
|
+
# wrong arity, a raised Thor::Error). Thor's legacy default is to print the
|
|
23
|
+
# error and exit **0**, which in CI reads as success: a caller passing a
|
|
24
|
+
# removed flag would write no site at all and still go green. The Pages
|
|
25
|
+
# deploy in relaton/support runs unattended across ~29 data repos, so a
|
|
26
|
+
# silent green there publishes an empty index nobody notices.
|
|
27
|
+
def self.exit_on_failure?
|
|
28
|
+
true
|
|
29
|
+
end
|
|
30
|
+
|
|
17
31
|
desc "version", "Show Relaton version"
|
|
18
32
|
|
|
19
33
|
def version
|
|
@@ -132,6 +146,71 @@ module Relaton
|
|
|
132
146
|
Relaton::Cli::YAMLConvertor.to_html(file, style, template)
|
|
133
147
|
end
|
|
134
148
|
|
|
149
|
+
desc "index [DATA-DIR]",
|
|
150
|
+
"Build a browsable HTML index site from a folder of Relaton YAML docs (default ./data)"
|
|
151
|
+
option :output, aliases: :o, default: "_site",
|
|
152
|
+
desc: "Output directory for the generated site"
|
|
153
|
+
option :flavor,
|
|
154
|
+
desc: "Build from relaton/relaton-data-<flavor> (shallow-cloned) instead of a local folder"
|
|
155
|
+
option :repo,
|
|
156
|
+
desc: "Build from a GitHub repo ORG/NAME (shallow-cloned) instead of a local folder"
|
|
157
|
+
option :branch, desc: "Branch to clone when --flavor/--repo is used"
|
|
158
|
+
option :title, aliases: :t, desc: "Index page title"
|
|
159
|
+
option :description,
|
|
160
|
+
desc: "Index page description (<meta name=\"description\"> and the header subtitle)"
|
|
161
|
+
option :favicon,
|
|
162
|
+
desc: "Favicon href for <link rel=\"icon\">, used verbatim (a relative " \
|
|
163
|
+
"path must be placed in the deployed site by the caller)"
|
|
164
|
+
option :"base-url",
|
|
165
|
+
desc: "Base URL for raw YAML links (defaults to the raw GitHub URL when cloning)"
|
|
166
|
+
option :"shard-size", type: :numeric, default: 5000,
|
|
167
|
+
desc: "Summary records per search-NNNN.json shard"
|
|
168
|
+
option :"detail-shard-size", type: :numeric, default: 500,
|
|
169
|
+
desc: "Detail records per detail-NNNN.json shard"
|
|
170
|
+
option :detail, type: :boolean, default: true,
|
|
171
|
+
desc: "Emit the detail shards the document panel fetches; --no-detail disables"
|
|
172
|
+
option :static, type: :boolean, default: true,
|
|
173
|
+
desc: "Also index a sibling static/ folder next to DATA-DIR (auto-detected); --no-static disables"
|
|
174
|
+
option :overwrite, aliases: :f, type: :boolean, default: true,
|
|
175
|
+
desc: "Overwrite existing output files"
|
|
176
|
+
option :"publish-data", type: :boolean, default: false,
|
|
177
|
+
desc: "Copy the corpus onto the site so document fetches " \
|
|
178
|
+
"share the Pages origin (for repos whose data is " \
|
|
179
|
+
"served only from there); --no-publish-data disables"
|
|
180
|
+
option :"pubid-flavor",
|
|
181
|
+
desc: "Flavor whose pubid Identifier parses docids and whose INDEXFILE " \
|
|
182
|
+
"names the published index (e.g. iso, iho, iec). Required unless " \
|
|
183
|
+
"--no-machine-index"
|
|
184
|
+
option :"machine-index", type: :boolean, default: true,
|
|
185
|
+
desc: "Emit the machine-consumable index (index/manifest.json, " \
|
|
186
|
+
"index/shard-NNNNN.json and the index-vN monolith); " \
|
|
187
|
+
"--no-machine-index builds the human site only"
|
|
188
|
+
option :"index-name",
|
|
189
|
+
desc: "Base name for the published index, overriding the flavor's " \
|
|
190
|
+
"INDEXFILE (e.g. index-v2). For a corpus that is not a relaton flavor"
|
|
191
|
+
|
|
192
|
+
def index(data_dir = nil)
|
|
193
|
+
with_index_source(data_dir) do |dir, default_base_url, default_title|
|
|
194
|
+
Relaton::Cli::IndexSiteGenerator.generate(
|
|
195
|
+
dir,
|
|
196
|
+
output: options[:output],
|
|
197
|
+
shard_size: options[:"shard-size"].to_i,
|
|
198
|
+
detail_shard_size: options[:"detail-shard-size"].to_i,
|
|
199
|
+
detail: options[:detail],
|
|
200
|
+
title: options[:title] || default_title,
|
|
201
|
+
description: options[:description],
|
|
202
|
+
favicon: options[:favicon],
|
|
203
|
+
base_url: options[:"base-url"] || default_base_url,
|
|
204
|
+
static: options[:static],
|
|
205
|
+
overwrite: options[:overwrite],
|
|
206
|
+
publish_data: options[:"publish-data"],
|
|
207
|
+
flavor: options[:"pubid-flavor"],
|
|
208
|
+
machine_index: options[:"machine-index"],
|
|
209
|
+
index_name: options[:"index-name"],
|
|
210
|
+
)
|
|
211
|
+
end
|
|
212
|
+
end
|
|
213
|
+
|
|
135
214
|
desc "convert XML", "Convert Relaton XML document"
|
|
136
215
|
option :format, aliases: :f, required: true, desc: "Output format (yaml, bibtex, asciibib)"
|
|
137
216
|
option :output, aliases: :o, desc: "Output to the specified file"
|
|
@@ -176,6 +255,41 @@ module Relaton
|
|
|
176
255
|
end
|
|
177
256
|
end
|
|
178
257
|
end
|
|
258
|
+
|
|
259
|
+
# Resolve the source folder for `index`. Locally it's just DATA-DIR
|
|
260
|
+
# (default ./data). With --flavor/--repo it shallow-clones the GitHub
|
|
261
|
+
# repo into a temp dir and yields its data/ subfolder plus sensible
|
|
262
|
+
# defaults for the raw-YAML base URL and page title. The temp clone is
|
|
263
|
+
# removed after the block returns.
|
|
264
|
+
def with_index_source(data_dir)
|
|
265
|
+
subdir = data_dir || "data"
|
|
266
|
+
if options[:flavor] || options[:repo]
|
|
267
|
+
repo = options[:repo] || "relaton/relaton-data-#{options[:flavor]}"
|
|
268
|
+
flavor = options[:flavor] ||
|
|
269
|
+
repo.split("/").last.sub(/\Arelaton-data-/, "")
|
|
270
|
+
Dir.mktmpdir("relaton-index-") do |tmp|
|
|
271
|
+
branch = clone_data_repo(repo, options[:branch], tmp)
|
|
272
|
+
base = "https://raw.githubusercontent.com/#{repo}/#{branch}"
|
|
273
|
+
yield File.join(tmp, subdir), base, "#{flavor.upcase} Index"
|
|
274
|
+
end
|
|
275
|
+
else
|
|
276
|
+
yield subdir, nil, nil
|
|
277
|
+
end
|
|
278
|
+
end
|
|
279
|
+
|
|
280
|
+
# Shallow-clone ORG/NAME into `dest`; return the checked-out branch name.
|
|
281
|
+
def clone_data_repo(repo, branch, dest)
|
|
282
|
+
url = "https://github.com/#{repo}.git"
|
|
283
|
+
cmd = ["git", "clone", "--depth", "1", "--single-branch"]
|
|
284
|
+
cmd += ["--branch", branch] if branch
|
|
285
|
+
cmd += [url, dest]
|
|
286
|
+
Relaton::Cli::Util.info "Cloning #{url}…"
|
|
287
|
+
unless system(*cmd, out: File::NULL, err: File::NULL)
|
|
288
|
+
raise Thor::Error,
|
|
289
|
+
"Failed to clone #{url}. Check the repo/branch name and that git is installed."
|
|
290
|
+
end
|
|
291
|
+
branch || `git -C #{Shellwords.escape(dest)} rev-parse --abbrev-ref HEAD`.strip
|
|
292
|
+
end
|
|
179
293
|
end
|
|
180
294
|
end
|
|
181
295
|
|
|
@@ -226,6 +340,8 @@ module Relaton
|
|
|
226
340
|
return "No matching bibliographic entry found" unless doc
|
|
227
341
|
|
|
228
342
|
serialize doc, options[:format]
|
|
343
|
+
rescue Pubid::Errors::Error
|
|
344
|
+
%("#{code}" is not a recognized standards identifier)
|
|
229
345
|
rescue Relaton::RequestError => e
|
|
230
346
|
e.message
|
|
231
347
|
end
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
module Relaton
|
|
2
|
+
module Cli
|
|
3
|
+
# Resolves the compiled frontend bundle (`app.iife.js` + `style.css`) that is
|
|
4
|
+
# built from `gems/relaton-cli/frontend/` at gem-build time and shipped inside
|
|
5
|
+
# the gem under `frontend/dist/`. Both the installed gem and a git checkout
|
|
6
|
+
# place `frontend/dist` at the gem root, so one relative path covers both.
|
|
7
|
+
#
|
|
8
|
+
# If the bundle is missing (a fresh checkout that hasn't run the frontend
|
|
9
|
+
# build), raise a clear, actionable error rather than emitting a broken page.
|
|
10
|
+
module FrontendAssets
|
|
11
|
+
class BuildMissingError < StandardError; end
|
|
12
|
+
|
|
13
|
+
# gem root = up three from lib/relaton/cli
|
|
14
|
+
DEFAULT_DIST = File.expand_path("../../../frontend/dist", __dir__).freeze
|
|
15
|
+
|
|
16
|
+
class << self
|
|
17
|
+
# Overridable so specs can point at a fixture dist without a Node build.
|
|
18
|
+
attr_writer :dist_dir
|
|
19
|
+
|
|
20
|
+
def dist_dir
|
|
21
|
+
@dist_dir || DEFAULT_DIST
|
|
22
|
+
end
|
|
23
|
+
|
|
24
|
+
# Run a block with a temporary dist dir, restoring the previous one.
|
|
25
|
+
def with_dist_dir(dir)
|
|
26
|
+
previous = @dist_dir
|
|
27
|
+
@dist_dir = dir
|
|
28
|
+
yield
|
|
29
|
+
ensure
|
|
30
|
+
@dist_dir = previous
|
|
31
|
+
end
|
|
32
|
+
|
|
33
|
+
def iife
|
|
34
|
+
read("app.iife.js")
|
|
35
|
+
end
|
|
36
|
+
|
|
37
|
+
def stylesheet
|
|
38
|
+
read("style.css")
|
|
39
|
+
end
|
|
40
|
+
|
|
41
|
+
def built?
|
|
42
|
+
File.exist?(File.join(dist_dir, "app.iife.js"))
|
|
43
|
+
end
|
|
44
|
+
|
|
45
|
+
def read(name)
|
|
46
|
+
path = File.join(dist_dir, name)
|
|
47
|
+
unless File.exist?(path)
|
|
48
|
+
raise BuildMissingError,
|
|
49
|
+
"Frontend build missing (#{name} not found in #{dist_dir}). " \
|
|
50
|
+
"Run `bundle exec rake build_frontend` in gems/relaton-cli " \
|
|
51
|
+
"(or `cd frontend && npm install && npm run build`) first."
|
|
52
|
+
end
|
|
53
|
+
File.read(path, encoding: "utf-8")
|
|
54
|
+
end
|
|
55
|
+
end
|
|
56
|
+
end
|
|
57
|
+
end
|
|
58
|
+
end
|
|
@@ -0,0 +1,380 @@
|
|
|
1
|
+
module Relaton
|
|
2
|
+
module Cli
|
|
3
|
+
# Turns a Relaton bibliographic YAML document (already parsed to a Hash) into
|
|
4
|
+
# the compact record the index frontend consumes. Works directly off the
|
|
5
|
+
# document's own rendered fields (docidentifier/title/date/ext.doctype), so no
|
|
6
|
+
# per-flavor pubid reconstruction is needed — the data files carry string
|
|
7
|
+
# DocIDs.
|
|
8
|
+
#
|
|
9
|
+
# Output record keys (shared contract with frontend/src/lib/types.ts):
|
|
10
|
+
# id, title, doctype, stage, date, link, yaml (always present) plus the
|
|
11
|
+
# optional detail-page fields (only when non-empty): abstract, edition,
|
|
12
|
+
# languages, keywords, publisher, contributors, docids, dates, relations.
|
|
13
|
+
module IndexItemNormalizer
|
|
14
|
+
module_function
|
|
15
|
+
|
|
16
|
+
# @param doc [Hash] parsed Relaton YAML (string keys)
|
|
17
|
+
# @param lang [String] preferred language for title/docid
|
|
18
|
+
# @param yaml_ref [String, nil] URL/path to the raw YAML source
|
|
19
|
+
# @return [Hash]
|
|
20
|
+
def normalize(doc, lang: "en", yaml_ref: nil)
|
|
21
|
+
{
|
|
22
|
+
"id" => docidentifier(doc, lang),
|
|
23
|
+
"title" => title(doc, lang),
|
|
24
|
+
"doctype" => doctype(doc),
|
|
25
|
+
"stage" => stage(doc),
|
|
26
|
+
"date" => date(doc),
|
|
27
|
+
"link" => link(doc, lang),
|
|
28
|
+
"yaml" => yaml_ref,
|
|
29
|
+
}.merge(details(doc, lang))
|
|
30
|
+
end
|
|
31
|
+
|
|
32
|
+
# Richer, structured fields the detail page renders. Empty/nil values are
|
|
33
|
+
# dropped so a summary-only doc keeps the compact core shape and the
|
|
34
|
+
# embedded JSON payload stays lean. These ride only in the embedded
|
|
35
|
+
# payload; search.json and the crawler DOM stay summary-only.
|
|
36
|
+
def details(doc, lang)
|
|
37
|
+
{
|
|
38
|
+
"abstract" => abstract(doc, lang),
|
|
39
|
+
"edition" => edition(doc),
|
|
40
|
+
"languages" => languages(doc),
|
|
41
|
+
"keywords" => keywords(doc),
|
|
42
|
+
"publisher" => publisher(doc, lang),
|
|
43
|
+
"contributors" => contributors(doc, lang),
|
|
44
|
+
"docids" => docids(doc),
|
|
45
|
+
"dates" => dates(doc),
|
|
46
|
+
"relations" => relations(doc, lang),
|
|
47
|
+
}.reject { |_, v| v.nil? || (v.respond_to?(:empty?) && v.empty?) }
|
|
48
|
+
end
|
|
49
|
+
|
|
50
|
+
# Pick the primary, preferred-language rendered DocID; fall back to
|
|
51
|
+
# docnumber / id. Values can be plain strings or {content, ...} hashes.
|
|
52
|
+
def docidentifier(doc, lang)
|
|
53
|
+
primary_docidentifier(doc, lang) || strip(doc["docnumber"]) || doc["id"].to_s
|
|
54
|
+
end
|
|
55
|
+
|
|
56
|
+
# Just the rendered primary DocID, with no docnumber/id fallback — nil when
|
|
57
|
+
# the item carries no docidentifier. Split out because a relation's target
|
|
58
|
+
# needs the DocID *only*: its `id` is an internal XML anchor, which must not
|
|
59
|
+
# outrank the formattedref (see `relation_id`).
|
|
60
|
+
def primary_docidentifier(doc, lang)
|
|
61
|
+
ids = wrap(doc["docidentifier"])
|
|
62
|
+
primary = ids.select { |d| d.is_a?(Hash) && truthy(d["primary"]) }
|
|
63
|
+
primary = ids if primary.empty?
|
|
64
|
+
|
|
65
|
+
chosen =
|
|
66
|
+
pick_lang(primary, lang) ||
|
|
67
|
+
primary.find { |d| d.is_a?(Hash) && d["language"].nil? } ||
|
|
68
|
+
primary.first
|
|
69
|
+
strip(content_of(chosen))
|
|
70
|
+
end
|
|
71
|
+
|
|
72
|
+
def title(doc, lang)
|
|
73
|
+
titles = wrap(doc["title"])
|
|
74
|
+
# A title entry may be {type: "main", ...} and/or language-tagged.
|
|
75
|
+
main = titles.select { |t| t.is_a?(Hash) && t["type"] == "main" }
|
|
76
|
+
pool = main.empty? ? titles : main
|
|
77
|
+
chosen = pick_lang(pool, lang) || pool.first
|
|
78
|
+
strip(content_of(chosen))
|
|
79
|
+
end
|
|
80
|
+
|
|
81
|
+
def doctype(doc)
|
|
82
|
+
dt = doc.dig("ext", "doctype") || doc["doctype"] || doc["type"]
|
|
83
|
+
strip(content_of(dt))
|
|
84
|
+
end
|
|
85
|
+
|
|
86
|
+
def stage(doc)
|
|
87
|
+
st = doc.dig("status", "stage") || doc["stage"]
|
|
88
|
+
strip(content_of(st))
|
|
89
|
+
end
|
|
90
|
+
|
|
91
|
+
# date: array of {type, at|on} or a plain string. Prefer "published".
|
|
92
|
+
def date(doc)
|
|
93
|
+
dates = doc["date"]
|
|
94
|
+
return normalize_date(dates) if dates.is_a?(String)
|
|
95
|
+
|
|
96
|
+
arr = wrap(dates)
|
|
97
|
+
chosen = arr.find { |d| d.is_a?(Hash) && d["type"] == "published" } ||
|
|
98
|
+
arr.find { |d| d.is_a?(Hash) } || arr.first
|
|
99
|
+
val = chosen.is_a?(Hash) ? (chosen["at"] || chosen["on"] || chosen["from"]) : chosen
|
|
100
|
+
normalize_date(val)
|
|
101
|
+
end
|
|
102
|
+
|
|
103
|
+
# Human-facing link: prefer a citation source in the preferred language.
|
|
104
|
+
def link(doc, lang)
|
|
105
|
+
sources = Array(doc["source"]) + Array(doc["link"])
|
|
106
|
+
return nil if sources.empty?
|
|
107
|
+
|
|
108
|
+
citation = sources.select do |s|
|
|
109
|
+
s.is_a?(Hash) && %w[citation web src].include?(s["type"])
|
|
110
|
+
end
|
|
111
|
+
pool = citation.empty? ? sources : citation
|
|
112
|
+
chosen = pick_lang(pool, lang) || pool.first
|
|
113
|
+
strip(content_of(chosen))
|
|
114
|
+
end
|
|
115
|
+
|
|
116
|
+
# --- detail-page fields --------------------------------------------------
|
|
117
|
+
|
|
118
|
+
# Preferred-language abstract (String or [{language, content}]), HTML-stripped.
|
|
119
|
+
def abstract(doc, lang)
|
|
120
|
+
value = doc["abstract"]
|
|
121
|
+
return strip(value) if value.is_a?(String)
|
|
122
|
+
|
|
123
|
+
entries = Array(value)
|
|
124
|
+
strip(content_of(pick_lang(entries, lang) || entries.first))
|
|
125
|
+
end
|
|
126
|
+
|
|
127
|
+
def edition(doc)
|
|
128
|
+
strip(content_of(doc["edition"]))
|
|
129
|
+
end
|
|
130
|
+
|
|
131
|
+
# The document's language codes, e.g. ["en", "fr"].
|
|
132
|
+
def languages(doc)
|
|
133
|
+
wrap(doc["language"]).map { |l| strip(content_of(l)) }.compact
|
|
134
|
+
end
|
|
135
|
+
|
|
136
|
+
def keywords(doc)
|
|
137
|
+
wrap(doc["keyword"]).map { |k| strip(content_of(k)) }.compact
|
|
138
|
+
end
|
|
139
|
+
|
|
140
|
+
# Name of the first publisher-role contributor, if any.
|
|
141
|
+
def publisher(doc, lang)
|
|
142
|
+
entry = wrap(doc["contributor"]).find { |c| role?(c, "publisher") }
|
|
143
|
+
entry && contributor_name(entry, lang)
|
|
144
|
+
end
|
|
145
|
+
|
|
146
|
+
# All contributors as {name, role}. Entries without a resolvable name are
|
|
147
|
+
# skipped so the detail page never shows a blank author line.
|
|
148
|
+
def contributors(doc, lang)
|
|
149
|
+
wrap(doc["contributor"]).filter_map { |c| contributor(c, lang) }
|
|
150
|
+
end
|
|
151
|
+
|
|
152
|
+
def contributor(entry, lang)
|
|
153
|
+
return nil unless entry.is_a?(Hash)
|
|
154
|
+
|
|
155
|
+
name = contributor_name(entry, lang)
|
|
156
|
+
return nil unless name
|
|
157
|
+
|
|
158
|
+
role = strip(content_of(roles_of(entry).first))
|
|
159
|
+
{ "name" => name, "role" => role }.reject { |_, v| v.nil? }
|
|
160
|
+
end
|
|
161
|
+
|
|
162
|
+
# Prefer an organization name, then a person name, then a bare "name".
|
|
163
|
+
def contributor_name(entry, lang)
|
|
164
|
+
org = entry["organization"]
|
|
165
|
+
return org_name(org, lang) if org.is_a?(Hash)
|
|
166
|
+
|
|
167
|
+
person = entry["person"]
|
|
168
|
+
return person_name(person, lang) if person.is_a?(Hash)
|
|
169
|
+
|
|
170
|
+
strip(content_of(entry["name"]))
|
|
171
|
+
end
|
|
172
|
+
|
|
173
|
+
def org_name(org, lang)
|
|
174
|
+
names = wrap(org["name"])
|
|
175
|
+
strip(content_of(pick_lang(names, lang) || names.first))
|
|
176
|
+
end
|
|
177
|
+
|
|
178
|
+
def person_name(person, lang)
|
|
179
|
+
name = person["name"]
|
|
180
|
+
return strip(content_of(name)) unless name.is_a?(Hash)
|
|
181
|
+
|
|
182
|
+
complete = name["completename"] || name["completeName"]
|
|
183
|
+
return strip(content_of(pick_lang(wrap(complete), lang) || complete)) if complete
|
|
184
|
+
|
|
185
|
+
given = strip(content_of(name["forename"] || name["given"]))
|
|
186
|
+
surname = strip(content_of(name["surname"]))
|
|
187
|
+
joined = [given, surname].compact.join(" ")
|
|
188
|
+
joined.empty? ? nil : joined
|
|
189
|
+
end
|
|
190
|
+
|
|
191
|
+
# role[].type for a contributor, as an array of raw values.
|
|
192
|
+
def roles_of(entry)
|
|
193
|
+
roles = entry["role"]
|
|
194
|
+
roles = [roles] unless roles.is_a?(Array)
|
|
195
|
+
roles.map { |r| r.is_a?(Hash) ? r["type"] : r }
|
|
196
|
+
end
|
|
197
|
+
|
|
198
|
+
def role?(entry, type)
|
|
199
|
+
return false unless entry.is_a?(Hash)
|
|
200
|
+
|
|
201
|
+
roles_of(entry).any? { |r| strip(content_of(r)) == type }
|
|
202
|
+
end
|
|
203
|
+
|
|
204
|
+
# Every docidentifier as {id, type?, language?} (blank ids dropped).
|
|
205
|
+
def docids(doc)
|
|
206
|
+
wrap(doc["docidentifier"]).filter_map do |d|
|
|
207
|
+
id = strip(content_of(d))
|
|
208
|
+
next nil unless id
|
|
209
|
+
|
|
210
|
+
hash = d.is_a?(Hash) ? d : {}
|
|
211
|
+
{ "id" => id,
|
|
212
|
+
"type" => strip(content_of(hash["type"])),
|
|
213
|
+
"language" => strip(content_of(hash["language"])) }
|
|
214
|
+
.reject { |_, v| v.nil? }
|
|
215
|
+
end
|
|
216
|
+
end
|
|
217
|
+
|
|
218
|
+
# Every date as {type?, value}; a plain string date becomes {value}.
|
|
219
|
+
def dates(doc)
|
|
220
|
+
value = doc["date"]
|
|
221
|
+
entries = value.is_a?(String) ? [value] : wrap(value)
|
|
222
|
+
entries.filter_map { |d| date_entry(d) }
|
|
223
|
+
end
|
|
224
|
+
|
|
225
|
+
def date_entry(entry)
|
|
226
|
+
unless entry.is_a?(Hash)
|
|
227
|
+
val = normalize_date(entry)
|
|
228
|
+
return val && { "value" => val }
|
|
229
|
+
end
|
|
230
|
+
|
|
231
|
+
val = normalize_date(entry["at"] || entry["on"] || entry["from"])
|
|
232
|
+
return nil unless val
|
|
233
|
+
|
|
234
|
+
{ "type" => strip(content_of(entry["type"])), "value" => val }
|
|
235
|
+
.reject { |_, v| v.nil? }
|
|
236
|
+
end
|
|
237
|
+
|
|
238
|
+
# Related documents as {type, id}, e.g. {"type" => "obsoletedBy",
|
|
239
|
+
# "id" => "ISO 29862:2024"}. The id is what the detail page matches against
|
|
240
|
+
# the rest of the corpus to decide whether the relation can be rendered as
|
|
241
|
+
# an intra-dataset link, so it must be rendered exactly as a document's own
|
|
242
|
+
# `id` is — hence the shared `docidentifier` helper below.
|
|
243
|
+
#
|
|
244
|
+
# Self-references are dropped (BIPM's Metrologia articles relate to
|
|
245
|
+
# themselves once per carrier, which would render as a link to the page you
|
|
246
|
+
# are already on) and identical {type, id} pairs are collapsed.
|
|
247
|
+
def relations(doc, lang)
|
|
248
|
+
own = docidentifier(doc, lang)
|
|
249
|
+
wrap(doc["relation"])
|
|
250
|
+
.filter_map { |r| relation(r, lang) }
|
|
251
|
+
.reject { |r| r["id"] == own }
|
|
252
|
+
.uniq
|
|
253
|
+
end
|
|
254
|
+
|
|
255
|
+
def relation(entry, lang)
|
|
256
|
+
return nil unless entry.is_a?(Hash)
|
|
257
|
+
|
|
258
|
+
id = relation_id(entry["bibitem"], lang)
|
|
259
|
+
return nil unless id
|
|
260
|
+
|
|
261
|
+
{ "type" => strip(content_of(entry["type"])), "id" => id }.reject { |_, v| v.nil? }
|
|
262
|
+
end
|
|
263
|
+
|
|
264
|
+
# A relation's bibitem carries the same docidentifier shape as a top-level
|
|
265
|
+
# document, so the primary-DocID logic is shared — that is what makes the
|
|
266
|
+
# emitted id string-identical to the dataset id it must match.
|
|
267
|
+
#
|
|
268
|
+
# Order matters. `docidentifier` is NOT reused whole: its last fallback is
|
|
269
|
+
# the item's `id`, which on a relation bibitem is an internal XML anchor
|
|
270
|
+
# ("CENISO/TS21003-7-2008/A1-2010", as CEN records carry). That anchor can
|
|
271
|
+
# never match a document in the corpus, so letting it outrank the rendered
|
|
272
|
+
# formattedref would strand a target that IS present as plain text under a
|
|
273
|
+
# mangled label. The anchor stays only as the last resort.
|
|
274
|
+
def relation_id(bibitem, lang)
|
|
275
|
+
return strip(content_of(bibitem)) unless bibitem.is_a?(Hash)
|
|
276
|
+
|
|
277
|
+
primary_docidentifier(bibitem, lang) ||
|
|
278
|
+
strip(content_of(bibitem["formattedref"])) ||
|
|
279
|
+
strip(bibitem["docnumber"]) ||
|
|
280
|
+
strip(bibitem["id"])
|
|
281
|
+
end
|
|
282
|
+
|
|
283
|
+
# --- helpers -------------------------------------------------------------
|
|
284
|
+
|
|
285
|
+
# Wrap a value as an array WITHOUT `Array()`'s Hash-to-pairs explosion: a
|
|
286
|
+
# bare Hash (a single unwrapped entry) becomes a one-element array, not a
|
|
287
|
+
# list of [key, value] pairs. RelatonBib normally emits arrays for these
|
|
288
|
+
# repeatable fields, but a hand-written YAML may carry a lone Hash.
|
|
289
|
+
def wrap(value)
|
|
290
|
+
case value
|
|
291
|
+
when nil then []
|
|
292
|
+
when Array then value
|
|
293
|
+
else [value]
|
|
294
|
+
end
|
|
295
|
+
end
|
|
296
|
+
|
|
297
|
+
def pick_lang(entries, lang)
|
|
298
|
+
entries.find do |e|
|
|
299
|
+
e.is_a?(Hash) && (e["language"] == lang || Array(e["language"]).include?(lang))
|
|
300
|
+
end
|
|
301
|
+
end
|
|
302
|
+
|
|
303
|
+
# Extract a display string from a value that may be a String or a Hash with
|
|
304
|
+
# a "content"/"name" key (recursing into arrays/hashes as needed).
|
|
305
|
+
def content_of(value)
|
|
306
|
+
case value
|
|
307
|
+
when String then value
|
|
308
|
+
when Numeric then value.to_s
|
|
309
|
+
when Hash then content_of(value["content"] || value["name"])
|
|
310
|
+
when Array then content_of(value.first)
|
|
311
|
+
end
|
|
312
|
+
end
|
|
313
|
+
|
|
314
|
+
def normalize_date(value)
|
|
315
|
+
return nil if value.nil?
|
|
316
|
+
|
|
317
|
+
s = value.to_s.strip
|
|
318
|
+
return nil if s.empty?
|
|
319
|
+
|
|
320
|
+
# Keep YYYY, YYYY-MM, or YYYY-MM-DD as-is (leading 10 chars).
|
|
321
|
+
s[0, 10]
|
|
322
|
+
end
|
|
323
|
+
|
|
324
|
+
# Strip inline HTML (e.g. <sup>) so records are clean for JSON/data-attrs.
|
|
325
|
+
#
|
|
326
|
+
# Counts angle brackets in one left-to-right pass instead of using a
|
|
327
|
+
# regex, because both regex spellings are wrong here:
|
|
328
|
+
#
|
|
329
|
+
# * `gsub(/<[^>]+>/, "")` matches ACROSS a nested `<`, so on
|
|
330
|
+
# "<sc<x>ript>" it removes the outer span "<sc<x>" and lets the
|
|
331
|
+
# fragments either side of the tag it walked over close back up into
|
|
332
|
+
# "ript>" (CodeQL rb/incomplete-multi-character-sanitization);
|
|
333
|
+
# * narrowing to `/<[^<>]*>/` matches innermost-first and fixes that,
|
|
334
|
+
# but then one pass no longer suffices and re-running to a fixed point
|
|
335
|
+
# is quadratic — each pass peels a single nesting level off the whole
|
|
336
|
+
# string. Measured at 0.7s for 8k chars of "<"*n + ">"*n, i.e. a real
|
|
337
|
+
# ReDoS traded for the theoretical one the scanner flagged.
|
|
338
|
+
#
|
|
339
|
+
# This is O(n) with no backtracking and nothing to re-scan. `pending`
|
|
340
|
+
# holds the span opened by the outermost unclosed `<` so that a bare "<"
|
|
341
|
+
# that never closes survives as literal text ("5 < 6") rather than
|
|
342
|
+
# swallowing the rest of the value; a bare ">" is likewise kept.
|
|
343
|
+
def strip(value)
|
|
344
|
+
return nil if value.nil?
|
|
345
|
+
|
|
346
|
+
s = strip_tags(value.to_s).strip
|
|
347
|
+
s.empty? ? nil : s
|
|
348
|
+
end
|
|
349
|
+
|
|
350
|
+
def strip_tags(str)
|
|
351
|
+
return str unless str.include?("<")
|
|
352
|
+
|
|
353
|
+
out = +""
|
|
354
|
+
pending = +""
|
|
355
|
+
depth = 0
|
|
356
|
+
|
|
357
|
+
str.each_char do |char|
|
|
358
|
+
if char == "<"
|
|
359
|
+
depth += 1
|
|
360
|
+
elsif depth.zero?
|
|
361
|
+
# Outside any tag — including a stray ">", which is not markup.
|
|
362
|
+
out << char
|
|
363
|
+
next
|
|
364
|
+
elsif char == ">"
|
|
365
|
+
depth -= 1
|
|
366
|
+
end
|
|
367
|
+
|
|
368
|
+
pending << char
|
|
369
|
+
pending.clear if depth.zero? # the span closed: drop it wholesale
|
|
370
|
+
end
|
|
371
|
+
|
|
372
|
+
depth.zero? ? out : out << pending
|
|
373
|
+
end
|
|
374
|
+
|
|
375
|
+
def truthy(value)
|
|
376
|
+
value == true || value == "true"
|
|
377
|
+
end
|
|
378
|
+
end
|
|
379
|
+
end
|
|
380
|
+
end
|