relaton-cli 3.0.0.pre.alpha.1 → 3.0.0.pre.alpha.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.gitignore +4 -0
- data/CLAUDE.md +403 -0
- data/Gemfile +5 -0
- data/Rakefile +15 -0
- data/docs/README.adoc +177 -1
- data/frontend/dist/404.html +18 -0
- data/frontend/dist/app.iife.js +1 -0
- data/frontend/dist/style.css +3 -0
- data/lib/relaton/cli/command.rb +116 -0
- data/lib/relaton/cli/frontend_assets.rb +58 -0
- data/lib/relaton/cli/index_item_normalizer.rb +380 -0
- data/lib/relaton/cli/index_site_generator.rb +860 -0
- data/lib/relaton/cli/subcommand_collection.rb +3 -0
- data/lib/relaton-cli.rb +5 -0
- data/relaton-cli.gemspec +13 -3
- data/templates/index/page.liquid +20 -0
- metadata +27 -9
|
@@ -0,0 +1,860 @@
|
|
|
1
|
+
require "json"
|
|
2
|
+
require "yaml"
|
|
3
|
+
require "zlib"
|
|
4
|
+
require "pathname"
|
|
5
|
+
require "fileutils"
|
|
6
|
+
require "liquid"
|
|
7
|
+
require "relaton/cli/frontend_assets"
|
|
8
|
+
require "relaton/index"
|
|
9
|
+
require "zip"
|
|
10
|
+
require "pubid"
|
|
11
|
+
require "relaton/cli/index_item_normalizer"
|
|
12
|
+
|
|
13
|
+
module Relaton
|
|
14
|
+
module Cli
|
|
15
|
+
# Builds a browsable HTML index site from a folder of Relaton bibliographic
|
|
16
|
+
# YAML documents (default ./data, arbitrarily nested).
|
|
17
|
+
#
|
|
18
|
+
# The page shell carries no document data at all — only branding, the
|
|
19
|
+
# inlined Vue+Tailwind bundle, and the scalars describing the shard layout.
|
|
20
|
+
# Documents are written as numbered JSON shards the frontend fetches:
|
|
21
|
+
#
|
|
22
|
+
# search-NNNN.json summary records {r,c,t,s,d,u,l}, loaded progressively
|
|
23
|
+
# detail-NNNN.json the rich fields, fetched when a detail panel opens
|
|
24
|
+
#
|
|
25
|
+
# So page weight is a function of what the user looks at, not of corpus size,
|
|
26
|
+
# and the generator streams rather than materializing the whole corpus.
|
|
27
|
+
class IndexSiteGenerator
|
|
28
|
+
TEMPLATE_DIR = File.expand_path("../../../templates/index", __dir__).freeze
|
|
29
|
+
# index YAMLs that are machine indexes, not documents.
|
|
30
|
+
SKIP_BASENAMES = /\Aindex(-v\d+)?\.ya?ml\z/i
|
|
31
|
+
# Sibling of the data folder holding manually-curated bib docs (ISO/IEC
|
|
32
|
+
# Directives, JCGM/GUM guides, NIST research-library metadata, …) that the
|
|
33
|
+
# crawler can't fetch. Part of the corpus (referenced by index-vN.yaml), so
|
|
34
|
+
# it belongs in the browsable site too.
|
|
35
|
+
STATIC_DIRNAME = "static".freeze
|
|
36
|
+
SHARD_PATTERN = "search-%04d.json".freeze
|
|
37
|
+
DETAIL_PATTERN = "detail-%04d.json".freeze
|
|
38
|
+
# Output written by a previous build that must not survive into this one:
|
|
39
|
+
# a shrunken corpus would otherwise leave orphan shards nothing points at.
|
|
40
|
+
STALE_GLOBS = ["search.json", "search-*.json", "detail-*.json", "index/*.json", "index-v*.yaml", "index-v*.zip"].freeze
|
|
41
|
+
# Normalized key -> compact key. Order defines the key order in a shard
|
|
42
|
+
# record; a key NOT listed here is a detail field (see #detail_record).
|
|
43
|
+
COMPACT_KEYS = {
|
|
44
|
+
"id" => "r", "title" => "c", "doctype" => "t", "stage" => "s",
|
|
45
|
+
"date" => "d", "yaml" => "u", "link" => "l"
|
|
46
|
+
}.freeze
|
|
47
|
+
# flavor token -> [pubid namespace, relaton namespace], for the tokens
|
|
48
|
+
# that do not follow the plain capitalize rule. 3GPP is the only one:
|
|
49
|
+
# `Pubid::Tgpp` against `Relaton::ThreeGpp`.
|
|
50
|
+
FLAVOR_NAMESPACES = {
|
|
51
|
+
"3gpp" => %w[Tgpp ThreeGpp], "tgpp" => %w[Tgpp ThreeGpp]
|
|
52
|
+
}.freeze
|
|
53
|
+
# A monolith base name: one path segment, so `--index-name` can neither
|
|
54
|
+
# write outside the output directory nor collide with a hidden file.
|
|
55
|
+
INDEX_NAME = /\A[A-Za-z0-9][A-Za-z0-9._-]*\z/
|
|
56
|
+
# <link rel="icon"> type hints, keyed by the favicon href's extension. An
|
|
57
|
+
# unlisted extension emits no type at all and lets the browser sniff.
|
|
58
|
+
FAVICON_TYPES = {
|
|
59
|
+
".svg" => "image/svg+xml", ".png" => "image/png", ".ico" => "image/x-icon",
|
|
60
|
+
".gif" => "image/gif", ".jpg" => "image/jpeg", ".jpeg" => "image/jpeg"
|
|
61
|
+
}.freeze
|
|
62
|
+
|
|
63
|
+
# Buffers records and flushes them as `dir/<pattern % n>` shards of at most
|
|
64
|
+
# `size` records, so peak memory is O(size), not O(corpus). Writing is
|
|
65
|
+
# delegated so the caller keeps control of the overwrite policy.
|
|
66
|
+
class ShardWriter
|
|
67
|
+
attr_reader :count
|
|
68
|
+
|
|
69
|
+
def initialize(dir, pattern, size, &write)
|
|
70
|
+
@dir = dir
|
|
71
|
+
@pattern = pattern
|
|
72
|
+
@size = size
|
|
73
|
+
@write = write
|
|
74
|
+
@buffer = []
|
|
75
|
+
@count = 0
|
|
76
|
+
end
|
|
77
|
+
|
|
78
|
+
def <<(record)
|
|
79
|
+
@buffer << record
|
|
80
|
+
flush if @buffer.size >= @size
|
|
81
|
+
self
|
|
82
|
+
end
|
|
83
|
+
|
|
84
|
+
# No-op on an empty buffer, so an exact multiple of `size` doesn't emit a
|
|
85
|
+
# trailing empty shard and an empty corpus emits none at all.
|
|
86
|
+
def flush
|
|
87
|
+
return if @buffer.empty?
|
|
88
|
+
|
|
89
|
+
@write.call(File.join(@dir, format(@pattern, @count)), JSON.generate(@buffer))
|
|
90
|
+
@count += 1
|
|
91
|
+
@buffer.clear
|
|
92
|
+
end
|
|
93
|
+
end
|
|
94
|
+
|
|
95
|
+
# The machine-consumable index: docid -> file rows a data repo
|
|
96
|
+
# publishes as index-vN.yaml, emitted on the Pages site as JSON
|
|
97
|
+
# shards plus a monolith, per the contract documented in
|
|
98
|
+
# relaton/relaton#113 (contract v2) and specified in
|
|
99
|
+
# docs/data-repository-format.adoc.
|
|
100
|
+
#
|
|
101
|
+
# Every row is structured: each data repo publishes a pubid index, so
|
|
102
|
+
# the generator always has a parser and never writes a plain-string id.
|
|
103
|
+
# A row the parser rejects takes its id from the repo's committed index
|
|
104
|
+
# (`committed:`), and is dropped only when that has no row for it either
|
|
105
|
+
# — `Relaton::Index::FileIO#deserialize_id` rejects the *whole* index on
|
|
106
|
+
# the first row it cannot deserialize, so one string row would poison
|
|
107
|
+
# the file.
|
|
108
|
+
#
|
|
109
|
+
# Shard key: `crc32(pubid.root.number.to_s) % N` — the same expression
|
|
110
|
+
# `Relaton::Index` bsearches on (`Type#candidates_by_number`,
|
|
111
|
+
# `FileIO#deserialize_pubid`). A document family (base, parts,
|
|
112
|
+
# amendments) shares one root and lands in one shard. The key is NOT
|
|
113
|
+
# given a rendered-id fallback: that would break the identity, so a
|
|
114
|
+
# client computing the key from its parsed query would look in a bucket
|
|
115
|
+
# the row is not in and read the miss as not-found. An identifier with
|
|
116
|
+
# no root number keys on "" and lands in shard 0.
|
|
117
|
+
class MachineIndex
|
|
118
|
+
TARGET_ROWS = 15
|
|
119
|
+
MIN_ROWS = 2000
|
|
120
|
+
MIN_SHARDS = 16
|
|
121
|
+
MAX_SHARDS = 65_536
|
|
122
|
+
|
|
123
|
+
# Deliberately does NOT retain the pubid object: `add` extracts the only
|
|
124
|
+
# two things the write path needs (`key`, `id_hash`) while the document
|
|
125
|
+
# streams past, then lets the identifier go. Retaining it costs 3.14 KB
|
|
126
|
+
# per row against 0.44 KB for the derived values alone — 543 MB vs 76 MB
|
|
127
|
+
# on a 177k-row corpus. Anything else derived from pubid must therefore
|
|
128
|
+
# be computed in `add` too; by write time the object is gone.
|
|
129
|
+
Row = Struct.new(:rendered, :file, :key, :id_hash)
|
|
130
|
+
|
|
131
|
+
attr_reader :rows
|
|
132
|
+
|
|
133
|
+
# @param pubid_class [Class] the flavor's pubid Identifier
|
|
134
|
+
# @param index_name [String] the monolith's base name, from the
|
|
135
|
+
# flavor's `INDEXFILE`
|
|
136
|
+
# @param committed [Hash, nil] `{ file => id hash }` read from the
|
|
137
|
+
# repo's own committed index, consulted only when a rendered docid
|
|
138
|
+
# does not parse
|
|
139
|
+
def initialize(pubid_class:, index_name:, committed: nil)
|
|
140
|
+
@pubid_class = pubid_class
|
|
141
|
+
@index_name = index_name
|
|
142
|
+
@committed = committed || {}
|
|
143
|
+
@rows = []
|
|
144
|
+
end
|
|
145
|
+
|
|
146
|
+
def add(rendered, file)
|
|
147
|
+
row = Row.new(rendered, file)
|
|
148
|
+
# Precompute the expensive derived values (root.number walk,
|
|
149
|
+
# lutaml to_hash) once during the streaming pass — computing
|
|
150
|
+
# them at monolith-write time is pathological on 177k rows
|
|
151
|
+
# because each involves object-graph traversal.
|
|
152
|
+
pubid = parse(rendered)
|
|
153
|
+
# A rendered docid the parser rejects falls back to the repo's own
|
|
154
|
+
# committed row: the crawler resolved that id from source metadata.
|
|
155
|
+
# The hash is kept verbatim, so the site's row is byte-identical to
|
|
156
|
+
# the repo's. A row with neither keeps a nil id_hash, which
|
|
157
|
+
# `indexed_rows` drops and `skipped_count` reports.
|
|
158
|
+
hash = pubid ? pubid.to_hash : @committed[file]
|
|
159
|
+
pubid ||= from_hash(hash)
|
|
160
|
+
if pubid
|
|
161
|
+
row.id_hash = hash
|
|
162
|
+
row.key = key_string(pubid)
|
|
163
|
+
end
|
|
164
|
+
@rows << row
|
|
165
|
+
end
|
|
166
|
+
|
|
167
|
+
def count
|
|
168
|
+
@rows.size
|
|
169
|
+
end
|
|
170
|
+
|
|
171
|
+
def key_strategy
|
|
172
|
+
"root-number"
|
|
173
|
+
end
|
|
174
|
+
|
|
175
|
+
# Sized from the rows actually written, the same figure the manifest
|
|
176
|
+
# reports as `count` — not from every row scanned.
|
|
177
|
+
def shard_count
|
|
178
|
+
written = indexed_rows.size
|
|
179
|
+
return 0 if written < MIN_ROWS
|
|
180
|
+
|
|
181
|
+
next_pow2((written.to_f / TARGET_ROWS).ceil).clamp(MIN_SHARDS, MAX_SHARDS)
|
|
182
|
+
end
|
|
183
|
+
|
|
184
|
+
def key_of(row)
|
|
185
|
+
row.key
|
|
186
|
+
end
|
|
187
|
+
|
|
188
|
+
def manifest(generated:)
|
|
189
|
+
{
|
|
190
|
+
"version" => 2,
|
|
191
|
+
# The monolith's base name, so a client can fetch
|
|
192
|
+
# "#{index}.zip" without knowing the flavor's INDEXFILE.
|
|
193
|
+
"index" => @index_name,
|
|
194
|
+
# What the index actually contains, not what was scanned: rows
|
|
195
|
+
# whose id could not be resolved at all are dropped.
|
|
196
|
+
"count" => indexed_rows.size,
|
|
197
|
+
"shards" => shard_count,
|
|
198
|
+
"key" => key_strategy,
|
|
199
|
+
"algorithm" => "crc32",
|
|
200
|
+
"generated" => generated,
|
|
201
|
+
}
|
|
202
|
+
end
|
|
203
|
+
|
|
204
|
+
# { "r" => rendered, "file" => path, "id" => pubid hash }. The same
|
|
205
|
+
# "id" the monolith carries for that row — the two artifacts must not
|
|
206
|
+
# describe one document differently.
|
|
207
|
+
def row_record(row)
|
|
208
|
+
{ "r" => row.rendered, "file" => row.file, "id" => row.id_hash }
|
|
209
|
+
end
|
|
210
|
+
|
|
211
|
+
# The rows actually written, in deterministic (key, rendered) order.
|
|
212
|
+
#
|
|
213
|
+
# Carries only rows whose id resolved: the consumer
|
|
214
|
+
# (`Relaton::Index::FileIO#deserialize_id`) calls `from_hash` on every
|
|
215
|
+
# row and raises `InvalidIndexError` on the first one it cannot
|
|
216
|
+
# deserialize, which rejects the *whole* index — so a single unresolved
|
|
217
|
+
# row written as a plain string would poison the file. Five shipping
|
|
218
|
+
# corpora carry a handful of ids that do not parse back from the
|
|
219
|
+
# rendered string (ieee 69, itu-r 47, iec 42, itu 3, nist 3), so this is
|
|
220
|
+
# reachable, not theoretical; `add` rescues those from the committed
|
|
221
|
+
# index, and only what that misses is dropped.
|
|
222
|
+
# `Relaton::Ieee::DataFetcher#build_index` already skips the same rows
|
|
223
|
+
# for the same reason and logs the loss; this matches it.
|
|
224
|
+
#
|
|
225
|
+
# Memoized, and the single place the sort happens — `each_shard` and
|
|
226
|
+
# `write_monolith` previously sorted the corpus independently, so both
|
|
227
|
+
# ran on every build.
|
|
228
|
+
def indexed_rows
|
|
229
|
+
@indexed_rows ||= @rows.reject { |row| row.id_hash.nil? }
|
|
230
|
+
.sort_by { |row| [row.key, row.rendered] }
|
|
231
|
+
end
|
|
232
|
+
|
|
233
|
+
# Rows dropped by `indexed_rows` — reported so the loss is never silent.
|
|
234
|
+
def skipped_count
|
|
235
|
+
count - indexed_rows.size
|
|
236
|
+
end
|
|
237
|
+
|
|
238
|
+
# Rows bucketed by shard, in deterministic (key, rendered) order;
|
|
239
|
+
# empty buckets are absent — a client treats a 404 as not-found.
|
|
240
|
+
def each_shard
|
|
241
|
+
n = shard_count
|
|
242
|
+
return enum_for(:each_shard) unless block_given? && n.positive?
|
|
243
|
+
|
|
244
|
+
buckets = Array.new(n) { [] }
|
|
245
|
+
indexed_rows.each { |row| buckets[Zlib.crc32(row.key) % n] << row_record(row) }
|
|
246
|
+
buckets.each_with_index { |rows, i| yield(format("%05d", i), rows) unless rows.empty? }
|
|
247
|
+
end
|
|
248
|
+
|
|
249
|
+
def monolith_filename
|
|
250
|
+
"#{@index_name}.yaml"
|
|
251
|
+
end
|
|
252
|
+
|
|
253
|
+
# Same shape Relaton::Index::FileIO#save emits: an Array of
|
|
254
|
+
# {id:, file:} hashes, id always a pubid to_hash. Streamed one row at a
|
|
255
|
+
# time — building the full array and calling to_yaml is
|
|
256
|
+
# pathological on six-figure corpora (Psych re-allocates on
|
|
257
|
+
# every nested hash, and the whole array sits in memory).
|
|
258
|
+
# Hand-rolls the YAML instead of calling to_yaml per row:
|
|
259
|
+
# Psych's per-invocation overhead on 177k rows takes ~40 min.
|
|
260
|
+
# The shapes are flat (string id) or one-level-nested (pubid
|
|
261
|
+
# to_hash), so hand-rendering is straightforward and the output
|
|
262
|
+
# is indistinguishable from what Psych produces.
|
|
263
|
+
def write_monolith(path)
|
|
264
|
+
File.open(path, "w:utf-8") do |f|
|
|
265
|
+
f << "---\n"
|
|
266
|
+
indexed_rows.each do |row|
|
|
267
|
+
f << "- :id:\n"
|
|
268
|
+
yaml_nested(f, row.id_hash, " ")
|
|
269
|
+
f << " :file: #{yaml_scalar(row.file)}\n"
|
|
270
|
+
end
|
|
271
|
+
end
|
|
272
|
+
end
|
|
273
|
+
|
|
274
|
+
# An Array is written as a JSON flow sequence, which is valid YAML and
|
|
275
|
+
# escapes every element unambiguously. It must not reach `yaml_scalar`:
|
|
276
|
+
# there `["IEC"].to_s` starts with "[", so it was quoted into the String
|
|
277
|
+
# "[\"IEC\"]" — every ISO/IEC copublished id (`copublishers`) — and
|
|
278
|
+
# `from_hash` cannot cast that, so `FileIO` rejects the whole index.
|
|
279
|
+
def yaml_nested(f, hash, indent)
|
|
280
|
+
hash.each do |k, v|
|
|
281
|
+
case v
|
|
282
|
+
when Hash
|
|
283
|
+
f << "#{indent}#{k}:\n"
|
|
284
|
+
yaml_nested(f, v, indent + " ")
|
|
285
|
+
when Array
|
|
286
|
+
f << "#{indent}#{k}: #{JSON.generate(v)}\n"
|
|
287
|
+
else
|
|
288
|
+
f << "#{indent}#{k}: #{yaml_scalar(v)}\n"
|
|
289
|
+
end
|
|
290
|
+
end
|
|
291
|
+
end
|
|
292
|
+
|
|
293
|
+
# YAML 1.1 plain scalars Psych resolves to true/false/nil. Emitted bare,
|
|
294
|
+
# an id or path with one of these literal values changes *type* on the
|
|
295
|
+
# round trip ("yes" -> true, "null" -> nil).
|
|
296
|
+
YAML11_PLAIN = /\A(?:y|Y|yes|Yes|YES|n|N|no|No|NO|true|True|TRUE|
|
|
297
|
+
false|False|FALSE|on|On|ON|off|Off|OFF|
|
|
298
|
+
null|Null|NULL|~)\z/x
|
|
299
|
+
|
|
300
|
+
def yaml_scalar(value)
|
|
301
|
+
# Booleans and numbers are emitted bare, so they keep their type.
|
|
302
|
+
# Quoting them is what the String branch below exists to prevent,
|
|
303
|
+
# inverted: a pubid `to_hash` carrying a real `true` (CIE's
|
|
304
|
+
# `d_prefix`, 31 of its 1139 rows) came back as the String "true", and
|
|
305
|
+
# the published row no longer matched the repo's own index.
|
|
306
|
+
return "" if value.nil?
|
|
307
|
+
return value.to_s if value == true || value == false || value.is_a?(Integer)
|
|
308
|
+
|
|
309
|
+
s = value.to_s
|
|
310
|
+
# Quote if the value could be ambiguous: leading special char, an
|
|
311
|
+
# embedded ": " or " #", numeric-looking, *trailing* whitespace (the
|
|
312
|
+
# guard here used to read /\z\s/, which can never match — nothing
|
|
313
|
+
# follows end-of-string), a YAML 1.1 plain word, or a C0 control
|
|
314
|
+
# character. The control case is the severe one: emitted raw it makes
|
|
315
|
+
# the whole file unparseable, and one bad row rejects the entire index.
|
|
316
|
+
if s.empty? ||
|
|
317
|
+
s.match?(/\A[-?:,\[\]{}#&*!|>'"%@`\s]|:\s|\s#|\A\d|\s\z/) ||
|
|
318
|
+
s.match?(YAML11_PLAIN) || s.match?(/[[:cntrl:]]/)
|
|
319
|
+
s.inspect
|
|
320
|
+
else
|
|
321
|
+
s
|
|
322
|
+
end
|
|
323
|
+
end
|
|
324
|
+
|
|
325
|
+
|
|
326
|
+
|
|
327
|
+
private
|
|
328
|
+
|
|
329
|
+
def next_pow2(value)
|
|
330
|
+
bit = 0
|
|
331
|
+
bit += 1 while (1 << bit) < value
|
|
332
|
+
1 << bit
|
|
333
|
+
end
|
|
334
|
+
|
|
335
|
+
def parse(rendered)
|
|
336
|
+
@pubid_class.parse(rendered)
|
|
337
|
+
rescue StandardError
|
|
338
|
+
nil
|
|
339
|
+
end
|
|
340
|
+
|
|
341
|
+
# Only a structured hash: a legacy index under the same name carries
|
|
342
|
+
# plain strings, and one written into a row would break the monolith
|
|
343
|
+
# (`yaml_nested` walks a Hash) and the consumer alike.
|
|
344
|
+
def from_hash(hash)
|
|
345
|
+
hash.is_a?(Hash) && @pubid_class.from_hash(hash) || nil
|
|
346
|
+
rescue StandardError
|
|
347
|
+
nil
|
|
348
|
+
end
|
|
349
|
+
|
|
350
|
+
# The narrowing key `Relaton::Index` sorts and bsearches on. An
|
|
351
|
+
# identifier with no root number gives "", so those rows cluster in
|
|
352
|
+
# shard 0 — the same degeneracy the gem's bsearch already has, and the
|
|
353
|
+
# only shape a client can reproduce without knowing our fallbacks.
|
|
354
|
+
def key_string(pubid)
|
|
355
|
+
pubid.root.number.to_s
|
|
356
|
+
end
|
|
357
|
+
end
|
|
358
|
+
|
|
359
|
+
# @param data_dir [String]
|
|
360
|
+
# @param options [Hash] :output :title :description :favicon :base_url
|
|
361
|
+
# :overwrite :lang :generated :static :shard_size :detail_shard_size
|
|
362
|
+
# :detail
|
|
363
|
+
def self.generate(data_dir, options = {})
|
|
364
|
+
new(data_dir, options).generate
|
|
365
|
+
end
|
|
366
|
+
|
|
367
|
+
def initialize(data_dir, options = {})
|
|
368
|
+
@data_dir = data_dir
|
|
369
|
+
@options = options
|
|
370
|
+
@output = options[:output] || "_site"
|
|
371
|
+
@lang = options[:lang] || "en"
|
|
372
|
+
@overwrite = options.fetch(:overwrite, true)
|
|
373
|
+
@base_url = options[:base_url]
|
|
374
|
+
@title = presence(options[:title]) || "Relaton Index"
|
|
375
|
+
@description = presence(options[:description])
|
|
376
|
+
@favicon = presence(options[:favicon])
|
|
377
|
+
@generated = options.fetch(:generated) { Time.now.utc.strftime("%Y-%m-%d") }
|
|
378
|
+
@include_static = options.fetch(:static, true)
|
|
379
|
+
@shard_size = options.fetch(:shard_size, 5000)
|
|
380
|
+
@detail_shard_size = options.fetch(:detail_shard_size, 500)
|
|
381
|
+
@emit_detail = options.fetch(:detail, true)
|
|
382
|
+
@emit_index = options.fetch(:machine_index, true)
|
|
383
|
+
@publish_data = options.fetch(:publish_data, false)
|
|
384
|
+
@pubid_class = pubid_class_for(options[:flavor])
|
|
385
|
+
@index_name = index_name_for(options[:flavor])
|
|
386
|
+
validate!
|
|
387
|
+
end
|
|
388
|
+
|
|
389
|
+
# @return [String] path to the written index.html
|
|
390
|
+
def generate
|
|
391
|
+
# Read the bundle first: a missing frontend build must fail before any
|
|
392
|
+
# output is written, not halfway through a corpus.
|
|
393
|
+
assets = { "css" => FrontendAssets.stylesheet, "iife" => FrontendAssets.iife }
|
|
394
|
+
|
|
395
|
+
FileUtils.mkdir_p(output)
|
|
396
|
+
purge_stale!
|
|
397
|
+
counts = write_shards
|
|
398
|
+
counts[:index_shards] = write_machine_index_manifest(counts[:total]) if emit_index?
|
|
399
|
+
publish_data! if @publish_data
|
|
400
|
+
|
|
401
|
+
index_path = File.join(output, "index.html")
|
|
402
|
+
write_file(index_path, render(assets, counts))
|
|
403
|
+
copy_404_fallback
|
|
404
|
+
Util.info "Indexed #{counts[:total]} document(s) from #{sources_description} " \
|
|
405
|
+
"into #{counts[:shards]} search shard(s), " \
|
|
406
|
+
"#{counts[:detail_shards]} detail shard(s) and " \
|
|
407
|
+
"#{counts[:index_shards]} machine-index shard(s)"
|
|
408
|
+
index_path
|
|
409
|
+
end
|
|
410
|
+
|
|
411
|
+
private
|
|
412
|
+
|
|
413
|
+
attr_reader :data_dir, :options, :output, :lang, :overwrite, :base_url,
|
|
414
|
+
:title, :description, :favicon, :generated, :include_static,
|
|
415
|
+
:shard_size, :detail_shard_size
|
|
416
|
+
|
|
417
|
+
def emit_detail?
|
|
418
|
+
@emit_detail
|
|
419
|
+
end
|
|
420
|
+
|
|
421
|
+
# `flavor` names the flavor ("iso", "iho", ...) whose pubid Identifier
|
|
422
|
+
# parses docids and whose `INDEXFILE` names the published monolith.
|
|
423
|
+
# Resolved from the two gems' own namespaces — the alias table carries
|
|
424
|
+
# only the names that do not follow the capitalize rule.
|
|
425
|
+
def pubid_class_for(flavor)
|
|
426
|
+
return nil unless flavor_key(flavor)
|
|
427
|
+
|
|
428
|
+
name = namespace_names(flavor).first
|
|
429
|
+
::Pubid.const_get(name).const_get(:Identifier)
|
|
430
|
+
rescue NameError => e
|
|
431
|
+
raise unless probed_constant?(e, name, :Identifier)
|
|
432
|
+
|
|
433
|
+
raise ArgumentError, "unknown pubid flavor: #{flavor}"
|
|
434
|
+
end
|
|
435
|
+
|
|
436
|
+
# The monolith's base name. Derived from the flavor's own `INDEXFILE`
|
|
437
|
+
# so the site publishes the file that flavor's consumer already fetches
|
|
438
|
+
# — the version there encodes the index *structure*, per flavor, and is
|
|
439
|
+
# not a global generation counter. `index_name:` overrides it for a
|
|
440
|
+
# corpus that is not a relaton flavor.
|
|
441
|
+
def index_name_for(flavor)
|
|
442
|
+
override = presence(options[:index_name])
|
|
443
|
+
return override if override
|
|
444
|
+
return nil unless flavor_key(flavor)
|
|
445
|
+
|
|
446
|
+
name = namespace_names(flavor).last
|
|
447
|
+
::Relaton.const_get(name).const_get(:INDEXFILE)
|
|
448
|
+
rescue NameError => e
|
|
449
|
+
raise unless probed_constant?(e, name, :INDEXFILE)
|
|
450
|
+
|
|
451
|
+
raise ArgumentError,
|
|
452
|
+
"no relaton flavor for `#{flavor}`; pass index_name to name the index"
|
|
453
|
+
end
|
|
454
|
+
|
|
455
|
+
# True only when the NameError is about the constant we looked up. Looking
|
|
456
|
+
# it up runs the flavor's autoload, and a genuine NameError from inside
|
|
457
|
+
# that file must surface as itself — relabelled "unknown flavor", it
|
|
458
|
+
# would send whoever reads a red deploy to --index-name instead of the bug.
|
|
459
|
+
def probed_constant?(error, *names)
|
|
460
|
+
names.map(&:to_s).include?(error.name.to_s)
|
|
461
|
+
end
|
|
462
|
+
|
|
463
|
+
def flavor_key(flavor)
|
|
464
|
+
key = flavor.to_s.strip.downcase
|
|
465
|
+
key unless key.empty?
|
|
466
|
+
end
|
|
467
|
+
|
|
468
|
+
# [pubid namespace, relaton namespace] for a flavor token. They agree for
|
|
469
|
+
# every flavor but 3GPP, whose pubid namespace is Tgpp.
|
|
470
|
+
def namespace_names(flavor)
|
|
471
|
+
key = flavor_key(flavor)
|
|
472
|
+
FLAVOR_NAMESPACES.fetch(key) do
|
|
473
|
+
name = key.split(/[_-]/).map(&:capitalize).join
|
|
474
|
+
[name, name]
|
|
475
|
+
end
|
|
476
|
+
end
|
|
477
|
+
|
|
478
|
+
def emit_index?
|
|
479
|
+
@emit_index
|
|
480
|
+
end
|
|
481
|
+
|
|
482
|
+
# The data folder's parent — the repo root the index rows' `file` paths
|
|
483
|
+
# and the committed index are relative to.
|
|
484
|
+
def repo_root
|
|
485
|
+
@repo_root ||= File.dirname(File.expand_path(data_dir))
|
|
486
|
+
end
|
|
487
|
+
|
|
488
|
+
# `{ file => id hash }` from the repo's own committed index, the
|
|
489
|
+
# authority for a docid the parser cannot read back from its rendered
|
|
490
|
+
# form. Empty when the repo publishes no index (a fresh repo, or one
|
|
491
|
+
# whose index this build is the first to produce).
|
|
492
|
+
def committed_index
|
|
493
|
+
return @committed_index if defined?(@committed_index)
|
|
494
|
+
|
|
495
|
+
path = File.join(repo_root, "#{@index_name}.yaml")
|
|
496
|
+
@committed_index = File.exist?(path) ? read_committed_index(path) : {}
|
|
497
|
+
rescue StandardError => e
|
|
498
|
+
Util.warn "Ignoring #{path}: #{e.message}"
|
|
499
|
+
@committed_index = {}
|
|
500
|
+
end
|
|
501
|
+
|
|
502
|
+
# Keeps only rows whose `:id` is a structured hash. A legacy index under
|
|
503
|
+
# the same name carries plain strings, and one of those written into a
|
|
504
|
+
# row would break both the monolith (`yaml_nested` walks a Hash) and the
|
|
505
|
+
# consumer (`FileIO#deserialize_id` calls `from_hash`).
|
|
506
|
+
def read_committed_index(path)
|
|
507
|
+
rows = YAML.safe_load(File.read(path), permitted_classes: [Symbol])
|
|
508
|
+
Array(rows).each_with_object({}) do |row, acc|
|
|
509
|
+
next unless row.is_a?(Hash) && row[:id].is_a?(Hash)
|
|
510
|
+
|
|
511
|
+
acc[row[:file]] = row[:id]
|
|
512
|
+
end
|
|
513
|
+
end
|
|
514
|
+
|
|
515
|
+
# nil for a nil/blank option value. A caller workflow that forwards an
|
|
516
|
+
# unset input renders it as an empty string (`--favicon ""`), which must
|
|
517
|
+
# mean "not set" — an empty href in <link rel="icon"> resolves to the page
|
|
518
|
+
# itself, and an empty <meta name="description"> is worse than none.
|
|
519
|
+
def presence(value)
|
|
520
|
+
str = value.to_s.strip
|
|
521
|
+
str unless str.empty?
|
|
522
|
+
end
|
|
523
|
+
|
|
524
|
+
def validate!
|
|
525
|
+
if options.key?(:mode)
|
|
526
|
+
raise ArgumentError,
|
|
527
|
+
"The `mode` option was removed: `relaton index` now always emits " \
|
|
528
|
+
"sharded JSON that the page fetches. Drop --mode from the call."
|
|
529
|
+
end
|
|
530
|
+
unless File.directory?(data_dir)
|
|
531
|
+
raise ArgumentError, "Data directory not found: #{data_dir}"
|
|
532
|
+
end
|
|
533
|
+
{ shard_size: shard_size, detail_shard_size: detail_shard_size }
|
|
534
|
+
.each do |name, value|
|
|
535
|
+
next if value.is_a?(Integer) && value.positive?
|
|
536
|
+
|
|
537
|
+
raise ArgumentError, "#{name} must be a positive integer (got #{value.inspect})"
|
|
538
|
+
end
|
|
539
|
+
|
|
540
|
+
# Every data repo publishes a pubid index, so a machine index without
|
|
541
|
+
# a parser could only carry plain-string ids — a shape no consumer
|
|
542
|
+
# narrows on and this generator no longer writes.
|
|
543
|
+
return unless emit_index?
|
|
544
|
+
|
|
545
|
+
if @pubid_class.nil?
|
|
546
|
+
raise ArgumentError,
|
|
547
|
+
"--pubid-flavor is required to build a machine index; " \
|
|
548
|
+
"pass --no-machine-index to build the human site only"
|
|
549
|
+
end
|
|
550
|
+
return if @index_name.match?(INDEX_NAME)
|
|
551
|
+
|
|
552
|
+
raise ArgumentError,
|
|
553
|
+
"index name must be a single file name such as index-v2 (got #{@index_name.inspect})"
|
|
554
|
+
end
|
|
555
|
+
|
|
556
|
+
# Human-readable description of the folders scanned, for the info log.
|
|
557
|
+
# Keeps data_dir as given (no absolute-path noise) and only notes when a
|
|
558
|
+
# sibling static/ was folded in.
|
|
559
|
+
def sources_description
|
|
560
|
+
repo_root = self.repo_root
|
|
561
|
+
static_source_dir(repo_root) ? "#{data_dir} (+ #{STATIC_DIRNAME}/)" : data_dir
|
|
562
|
+
end
|
|
563
|
+
|
|
564
|
+
# ---- streaming ----------------------------------------------------------
|
|
565
|
+
|
|
566
|
+
# The machine-index input: rendered primary docid + repo-relative
|
|
567
|
+
# path (clients prefix their own baseurl onto file).
|
|
568
|
+
def machine_record(item)
|
|
569
|
+
[item["id"], item["yaml_path"]]
|
|
570
|
+
end
|
|
571
|
+
|
|
572
|
+
def write_machine_index_manifest(total)
|
|
573
|
+
machine = @machine_index or return 0
|
|
574
|
+
|
|
575
|
+
FileUtils.mkdir_p(File.join(output, "index"))
|
|
576
|
+
machine.each_shard do |number, rows|
|
|
577
|
+
write_file(File.join(output, "index", "shard-#{number}.json"), JSON.generate(rows))
|
|
578
|
+
end
|
|
579
|
+
|
|
580
|
+
monolith = File.join(output, machine.monolith_filename)
|
|
581
|
+
machine.write_monolith(monolith)
|
|
582
|
+
zip_file(monolith)
|
|
583
|
+
|
|
584
|
+
write_file(
|
|
585
|
+
File.join(output, "index", "manifest.json"),
|
|
586
|
+
JSON.pretty_generate(machine.manifest(generated: @generated)),
|
|
587
|
+
)
|
|
588
|
+
if machine.skipped_count.positive?
|
|
589
|
+
Util.warn "Machine index: skipped #{machine.skipped_count} of " \
|
|
590
|
+
"#{machine.count} document(s) whose id could not be parsed " \
|
|
591
|
+
"(a structured index cannot carry them)."
|
|
592
|
+
end
|
|
593
|
+
machine.shard_count
|
|
594
|
+
end
|
|
595
|
+
|
|
596
|
+
# Zip sibling of a monolith — the form `Relaton::Index url:`
|
|
597
|
+
# fetches today. The zip stores the yaml at its bare basename,
|
|
598
|
+
# matching the layout data repos publish in git.
|
|
599
|
+
def zip_file(yaml_path)
|
|
600
|
+
zip_path = yaml_path.sub(/\.yaml\z/, ".zip")
|
|
601
|
+
File.delete(zip_path) if File.exist?(zip_path)
|
|
602
|
+
Zip::File.open(zip_path, create: true) do |archive|
|
|
603
|
+
archive.add(File.basename(yaml_path), yaml_path)
|
|
604
|
+
end
|
|
605
|
+
end
|
|
606
|
+
|
|
607
|
+
# Copy the scanned corpus onto the site so clients can fetch
|
|
608
|
+
# documents (the index rows' `file` paths) from the same origin.
|
|
609
|
+
# Opt-in: for most repos raw.githubusercontent already serves the
|
|
610
|
+
# committed data, and duplicating a large corpus would double the
|
|
611
|
+
# published-site size against the 1 GB cap.
|
|
612
|
+
def publish_data!
|
|
613
|
+
repo_root = self.repo_root
|
|
614
|
+
source_dirs(repo_root).each do |dir|
|
|
615
|
+
Dir.glob(File.join(dir, "**", "*.{yaml,yml}")).sort.each do |src|
|
|
616
|
+
# Relative to the data dir itself, so the copy lives at
|
|
617
|
+
# output/data/<name>.yaml matching the index rows' file paths.
|
|
618
|
+
rel = Pathname.new(File.expand_path(src))
|
|
619
|
+
.relative_path_from(Pathname.new(dir)).to_s
|
|
620
|
+
dest = File.join(output, "data", rel)
|
|
621
|
+
FileUtils.mkdir_p(File.dirname(dest))
|
|
622
|
+
FileUtils.cp(src, dest)
|
|
623
|
+
end
|
|
624
|
+
end
|
|
625
|
+
end
|
|
626
|
+
|
|
627
|
+
# One pass over the corpus, fanning each document out to both shard
|
|
628
|
+
# families. The search/detail families accumulate nothing — peak memory is
|
|
629
|
+
# one shard of each. The machine index is the exception: shard assignment
|
|
630
|
+
# needs the corpus size, which is not knowable until the pass ends, so
|
|
631
|
+
# `MachineIndex` retains one `Row` per document (measured ~0.44 KB/row — 76 MB at 177k
|
|
632
|
+
# rows; see the note on `Row`, which is why it does not hold the pubid).
|
|
633
|
+
def write_shards
|
|
634
|
+
writer = method(:write_file)
|
|
635
|
+
summary = ShardWriter.new(output, SHARD_PATTERN, shard_size, &writer)
|
|
636
|
+
detail = ShardWriter.new(output, DETAIL_PATTERN, detail_shard_size, &writer)
|
|
637
|
+
if emit_index?
|
|
638
|
+
machine = MachineIndex.new(pubid_class: @pubid_class,
|
|
639
|
+
index_name: @index_name,
|
|
640
|
+
committed: committed_index)
|
|
641
|
+
end
|
|
642
|
+
total = 0
|
|
643
|
+
|
|
644
|
+
each_document do |doc|
|
|
645
|
+
total += 1
|
|
646
|
+
summary << compact_record(doc)
|
|
647
|
+
detail << detail_record(doc) if emit_detail?
|
|
648
|
+
machine&.add(*machine_record(doc))
|
|
649
|
+
end
|
|
650
|
+
summary.flush
|
|
651
|
+
detail.flush
|
|
652
|
+
@machine_index = machine
|
|
653
|
+
|
|
654
|
+
{ total: total, shards: summary.count, detail_shards: detail.count }
|
|
655
|
+
end
|
|
656
|
+
|
|
657
|
+
# Yield each index item in corpus order. De-dup is **cross-dir only**: an id
|
|
658
|
+
# already indexed from an *earlier* dir (i.e. static/ duplicating a data/
|
|
659
|
+
# doc) is skipped, so data/ wins — but duplicates *within* a single dir are
|
|
660
|
+
# left as-is, preserving the pre-static behavior of the data scan.
|
|
661
|
+
def each_document
|
|
662
|
+
return to_enum(:each_document) unless block_given?
|
|
663
|
+
|
|
664
|
+
repo_root = self.repo_root
|
|
665
|
+
seen = {}
|
|
666
|
+
source_dirs(repo_root).each do |dir|
|
|
667
|
+
dir_ids = {}
|
|
668
|
+
Dir.glob(File.join(dir, "**", "*.{yaml,yml}")).sort.each do |file|
|
|
669
|
+
item = index_file(file, repo_root)
|
|
670
|
+
next unless item
|
|
671
|
+
|
|
672
|
+
id = dedup_key(item)
|
|
673
|
+
if id && seen.key?(id)
|
|
674
|
+
Util.warn "Skipping #{item['yaml']} (duplicate id #{id}); " \
|
|
675
|
+
"already indexed from #{seen[id]}"
|
|
676
|
+
next
|
|
677
|
+
end
|
|
678
|
+
dir_ids[id] ||= item["yaml"] if id
|
|
679
|
+
# The machine index needs the repo-relative path: clients
|
|
680
|
+
# prefix their own baseurl onto row[:file], so an absolutized
|
|
681
|
+
# ref (yaml_ref with --base-url) would double-prefix.
|
|
682
|
+
item["yaml_path"] = relative_path(file, repo_root)
|
|
683
|
+
yield item
|
|
684
|
+
end
|
|
685
|
+
seen.merge!(dir_ids)
|
|
686
|
+
end
|
|
687
|
+
end
|
|
688
|
+
|
|
689
|
+
# The compact summary record the search shards carry.
|
|
690
|
+
def compact_record(item)
|
|
691
|
+
COMPACT_KEYS.each_with_object({}) { |(key, short), acc| acc[short] = item[key] }
|
|
692
|
+
end
|
|
693
|
+
|
|
694
|
+
# The complement: everything the summary record doesn't carry, tagged with
|
|
695
|
+
# the id so the frontend can verify a positional lookup landed on the right
|
|
696
|
+
# record. nil when the document has no detail fields at all — the slot is
|
|
697
|
+
# still written (as null) so position stays aligned with the summary shards.
|
|
698
|
+
def detail_record(item)
|
|
699
|
+
# "yaml_path" is transport for the machine index, not a detail field.
|
|
700
|
+
rest = item.reject { |key, _| COMPACT_KEYS.key?(key) || key == "yaml_path" }
|
|
701
|
+
return nil if rest.empty?
|
|
702
|
+
|
|
703
|
+
{ "r" => item["id"] }.merge(rest)
|
|
704
|
+
end
|
|
705
|
+
|
|
706
|
+
# The data folder, plus an auto-detected sibling static/ folder (its bib
|
|
707
|
+
# docs are part of the corpus). Data is scanned first so it wins on a
|
|
708
|
+
# cross-dir duplicate id. Enabled by default; --no-static opts out.
|
|
709
|
+
def source_dirs(repo_root)
|
|
710
|
+
[data_dir, static_source_dir(repo_root)].compact
|
|
711
|
+
end
|
|
712
|
+
|
|
713
|
+
# The sibling static/ dir to fold in, or nil when disabled, absent, or the
|
|
714
|
+
# same folder as data_dir (guards `relaton index static` double-scanning).
|
|
715
|
+
def static_source_dir(repo_root)
|
|
716
|
+
return nil unless include_static
|
|
717
|
+
|
|
718
|
+
static = File.join(repo_root, STATIC_DIRNAME)
|
|
719
|
+
return nil unless File.directory?(static)
|
|
720
|
+
return nil if File.expand_path(static) == File.expand_path(data_dir)
|
|
721
|
+
|
|
722
|
+
static
|
|
723
|
+
end
|
|
724
|
+
|
|
725
|
+
# The id we de-dup on, or nil when the doc has no usable id (a blank/empty
|
|
726
|
+
# id must NOT collapse distinct docid-less documents together).
|
|
727
|
+
def dedup_key(item)
|
|
728
|
+
id = item["id"]
|
|
729
|
+
id unless id.nil? || id.empty?
|
|
730
|
+
end
|
|
731
|
+
|
|
732
|
+
# Normalize one YAML file to an index item, or nil if it's a machine index,
|
|
733
|
+
# not a document, or unparseable.
|
|
734
|
+
def index_file(file, repo_root)
|
|
735
|
+
return if File.basename(file).match?(SKIP_BASENAMES)
|
|
736
|
+
|
|
737
|
+
doc = load_yaml(file)
|
|
738
|
+
return unless document?(doc)
|
|
739
|
+
|
|
740
|
+
rel = relative_path(file, repo_root)
|
|
741
|
+
IndexItemNormalizer.normalize(doc, lang: lang, yaml_ref: yaml_ref(rel))
|
|
742
|
+
rescue Psych::SyntaxError => e
|
|
743
|
+
Util.warn "Skipping #{file}: #{e.message}"
|
|
744
|
+
nil
|
|
745
|
+
end
|
|
746
|
+
|
|
747
|
+
def load_yaml(file)
|
|
748
|
+
content = File.read(file, encoding: "utf-8")
|
|
749
|
+
begin
|
|
750
|
+
YAML.safe_load(content, permitted_classes: [Date, Time], aliases: true)
|
|
751
|
+
rescue ArgumentError
|
|
752
|
+
# older Psych positional signature
|
|
753
|
+
YAML.safe_load(content, [Date, Time], [], true)
|
|
754
|
+
end
|
|
755
|
+
end
|
|
756
|
+
|
|
757
|
+
# A bib document (not a collection/index) — has an id/docid/title, no "root".
|
|
758
|
+
def document?(doc)
|
|
759
|
+
return false unless doc.is_a?(Hash)
|
|
760
|
+
return false if doc.key?("root")
|
|
761
|
+
|
|
762
|
+
doc.key?("id") || doc.key?("docidentifier") || doc.key?("title")
|
|
763
|
+
end
|
|
764
|
+
|
|
765
|
+
def relative_path(file, repo_root)
|
|
766
|
+
Pathname.new(File.expand_path(file))
|
|
767
|
+
.relative_path_from(Pathname.new(repo_root)).to_s
|
|
768
|
+
end
|
|
769
|
+
|
|
770
|
+
def yaml_ref(rel)
|
|
771
|
+
return rel if base_url.nil? || base_url.empty?
|
|
772
|
+
|
|
773
|
+
"#{base_url.chomp('/')}/#{rel}"
|
|
774
|
+
end
|
|
775
|
+
|
|
776
|
+
# ---- rendering ----------------------------------------------------------
|
|
777
|
+
|
|
778
|
+
def render(assets, counts)
|
|
779
|
+
env = Liquid::Environment.build(
|
|
780
|
+
file_system: Liquid::LocalFileSystem.new(TEMPLATE_DIR),
|
|
781
|
+
)
|
|
782
|
+
template = Liquid::Template.parse(
|
|
783
|
+
File.read(File.join(TEMPLATE_DIR, "page.liquid"), encoding: "utf-8"),
|
|
784
|
+
environment: env,
|
|
785
|
+
)
|
|
786
|
+
template.render!(
|
|
787
|
+
"title" => title,
|
|
788
|
+
"description" => description,
|
|
789
|
+
"favicon" => favicon,
|
|
790
|
+
"favicon_type" => favicon_type,
|
|
791
|
+
"css" => assets["css"],
|
|
792
|
+
"iife" => assets["iife"],
|
|
793
|
+
"generated" => generated,
|
|
794
|
+
"total" => counts[:total],
|
|
795
|
+
"shards" => counts[:shards],
|
|
796
|
+
"shard_size" => shard_size,
|
|
797
|
+
"detail_shards" => counts[:detail_shards],
|
|
798
|
+
"detail_shard_size" => detail_shard_size,
|
|
799
|
+
)
|
|
800
|
+
end
|
|
801
|
+
|
|
802
|
+
# The <link rel="icon"> type for the configured favicon, or nil when there
|
|
803
|
+
# is none or its extension isn't a known image type. The href is passed
|
|
804
|
+
# through verbatim (absolute URL or output-relative path alike), so any
|
|
805
|
+
# ?query/#fragment is stripped before looking at the extension.
|
|
806
|
+
def favicon_type
|
|
807
|
+
return nil unless favicon
|
|
808
|
+
|
|
809
|
+
FAVICON_TYPES[File.extname(favicon.split(/[?#]/, 2).first.to_s).downcase]
|
|
810
|
+
end
|
|
811
|
+
|
|
812
|
+
# Drop shards from a previous build. Without a manifest, an orphan shard
|
|
813
|
+
# left by a larger corpus is invisible — nothing points at it — so it would
|
|
814
|
+
# sit in the deployed site indefinitely.
|
|
815
|
+
# Path-based document URLs (/doc/<id>) are served by this fallback:
|
|
816
|
+
# GitHub Pages has no rewrites, so unknown paths resolve to 404.html,
|
|
817
|
+
# whose script forwards the id into the app as ?doc=<id>.
|
|
818
|
+
def copy_404_fallback
|
|
819
|
+
fallback = File.join(FrontendAssets.dist_dir, "404.html")
|
|
820
|
+
return unless File.exist?(fallback)
|
|
821
|
+
|
|
822
|
+
write_file(File.join(output, "404.html"), File.read(fallback))
|
|
823
|
+
end
|
|
824
|
+
|
|
825
|
+
def purge_stale!
|
|
826
|
+
return unless overwrite
|
|
827
|
+
|
|
828
|
+
# Read before the glob deletes it. `--index-name` accepts a name
|
|
829
|
+
# `index-v*` does not match, so the previous build's own manifest is
|
|
830
|
+
# the only record of which monolith it wrote.
|
|
831
|
+
previous = previous_index_name
|
|
832
|
+
# `base:` rather than interpolating `output` into the pattern: an output
|
|
833
|
+
# path containing glob metacharacters (`[`, `{`, `*`, `?`, …) would
|
|
834
|
+
# otherwise match nothing, and the stale shards would survive silently.
|
|
835
|
+
paths = Dir.glob(STALE_GLOBS, base: output).map { |name| File.join(output, name) }
|
|
836
|
+
paths += %w[yaml zip].map { |ext| File.join(output, "#{previous}.#{ext}") } if previous
|
|
837
|
+
paths.uniq.each { |path| File.delete(path) if File.file?(path) }
|
|
838
|
+
end
|
|
839
|
+
|
|
840
|
+
# The monolith name a previous build recorded, if it is a safe basename.
|
|
841
|
+
def previous_index_name
|
|
842
|
+
path = File.join(output, "index", "manifest.json")
|
|
843
|
+
return unless File.file?(path)
|
|
844
|
+
|
|
845
|
+
name = JSON.parse(File.read(path))["index"]
|
|
846
|
+
name if name.is_a?(String) && name.match?(INDEX_NAME)
|
|
847
|
+
rescue JSON::ParserError
|
|
848
|
+
nil
|
|
849
|
+
end
|
|
850
|
+
|
|
851
|
+
def write_file(path, content)
|
|
852
|
+
if File.exist?(path) && !overwrite
|
|
853
|
+
Util.warn "Skipping existing #{path} (use --overwrite)"
|
|
854
|
+
return
|
|
855
|
+
end
|
|
856
|
+
File.write(path, content, encoding: "utf-8")
|
|
857
|
+
end
|
|
858
|
+
end
|
|
859
|
+
end
|
|
860
|
+
end
|