relaton-cli 3.0.0.pre.alpha.1 → 3.0.0.pre.alpha.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,860 @@
1
+ require "json"
2
+ require "yaml"
3
+ require "zlib"
4
+ require "pathname"
5
+ require "fileutils"
6
+ require "liquid"
7
+ require "relaton/cli/frontend_assets"
8
+ require "relaton/index"
9
+ require "zip"
10
+ require "pubid"
11
+ require "relaton/cli/index_item_normalizer"
12
+
13
+ module Relaton
14
+ module Cli
15
+ # Builds a browsable HTML index site from a folder of Relaton bibliographic
16
+ # YAML documents (default ./data, arbitrarily nested).
17
+ #
18
+ # The page shell carries no document data at all — only branding, the
19
+ # inlined Vue+Tailwind bundle, and the scalars describing the shard layout.
20
+ # Documents are written as numbered JSON shards the frontend fetches:
21
+ #
22
+ # search-NNNN.json summary records {r,c,t,s,d,u,l}, loaded progressively
23
+ # detail-NNNN.json the rich fields, fetched when a detail panel opens
24
+ #
25
+ # So page weight is a function of what the user looks at, not of corpus size,
26
+ # and the generator streams rather than materializing the whole corpus.
27
+ class IndexSiteGenerator
28
+ TEMPLATE_DIR = File.expand_path("../../../templates/index", __dir__).freeze
29
+ # index YAMLs that are machine indexes, not documents.
30
+ SKIP_BASENAMES = /\Aindex(-v\d+)?\.ya?ml\z/i
31
+ # Sibling of the data folder holding manually-curated bib docs (ISO/IEC
32
+ # Directives, JCGM/GUM guides, NIST research-library metadata, …) that the
33
+ # crawler can't fetch. Part of the corpus (referenced by index-vN.yaml), so
34
+ # it belongs in the browsable site too.
35
+ STATIC_DIRNAME = "static".freeze
36
+ SHARD_PATTERN = "search-%04d.json".freeze
37
+ DETAIL_PATTERN = "detail-%04d.json".freeze
38
+ # Output written by a previous build that must not survive into this one:
39
+ # a shrunken corpus would otherwise leave orphan shards nothing points at.
40
+ STALE_GLOBS = ["search.json", "search-*.json", "detail-*.json", "index/*.json", "index-v*.yaml", "index-v*.zip"].freeze
41
+ # Normalized key -> compact key. Order defines the key order in a shard
42
+ # record; a key NOT listed here is a detail field (see #detail_record).
43
+ COMPACT_KEYS = {
44
+ "id" => "r", "title" => "c", "doctype" => "t", "stage" => "s",
45
+ "date" => "d", "yaml" => "u", "link" => "l"
46
+ }.freeze
47
+ # flavor token -> [pubid namespace, relaton namespace], for the tokens
48
+ # that do not follow the plain capitalize rule. 3GPP is the only one:
49
+ # `Pubid::Tgpp` against `Relaton::ThreeGpp`.
50
+ FLAVOR_NAMESPACES = {
51
+ "3gpp" => %w[Tgpp ThreeGpp], "tgpp" => %w[Tgpp ThreeGpp]
52
+ }.freeze
53
+ # A monolith base name: one path segment, so `--index-name` can neither
54
+ # write outside the output directory nor collide with a hidden file.
55
+ INDEX_NAME = /\A[A-Za-z0-9][A-Za-z0-9._-]*\z/
56
+ # <link rel="icon"> type hints, keyed by the favicon href's extension. An
57
+ # unlisted extension emits no type at all and lets the browser sniff.
58
+ FAVICON_TYPES = {
59
+ ".svg" => "image/svg+xml", ".png" => "image/png", ".ico" => "image/x-icon",
60
+ ".gif" => "image/gif", ".jpg" => "image/jpeg", ".jpeg" => "image/jpeg"
61
+ }.freeze
62
+
63
+ # Buffers records and flushes them as `dir/<pattern % n>` shards of at most
64
+ # `size` records, so peak memory is O(size), not O(corpus). Writing is
65
+ # delegated so the caller keeps control of the overwrite policy.
66
+ class ShardWriter
67
+ attr_reader :count
68
+
69
+ def initialize(dir, pattern, size, &write)
70
+ @dir = dir
71
+ @pattern = pattern
72
+ @size = size
73
+ @write = write
74
+ @buffer = []
75
+ @count = 0
76
+ end
77
+
78
+ def <<(record)
79
+ @buffer << record
80
+ flush if @buffer.size >= @size
81
+ self
82
+ end
83
+
84
+ # No-op on an empty buffer, so an exact multiple of `size` doesn't emit a
85
+ # trailing empty shard and an empty corpus emits none at all.
86
+ def flush
87
+ return if @buffer.empty?
88
+
89
+ @write.call(File.join(@dir, format(@pattern, @count)), JSON.generate(@buffer))
90
+ @count += 1
91
+ @buffer.clear
92
+ end
93
+ end
94
+
95
+ # The machine-consumable index: docid -> file rows a data repo
96
+ # publishes as index-vN.yaml, emitted on the Pages site as JSON
97
+ # shards plus a monolith, per the contract documented in
98
+ # relaton/relaton#113 (contract v2) and specified in
99
+ # docs/data-repository-format.adoc.
100
+ #
101
+ # Every row is structured: each data repo publishes a pubid index, so
102
+ # the generator always has a parser and never writes a plain-string id.
103
+ # A row the parser rejects takes its id from the repo's committed index
104
+ # (`committed:`), and is dropped only when that has no row for it either
105
+ # — `Relaton::Index::FileIO#deserialize_id` rejects the *whole* index on
106
+ # the first row it cannot deserialize, so one string row would poison
107
+ # the file.
108
+ #
109
+ # Shard key: `crc32(pubid.root.number.to_s) % N` — the same expression
110
+ # `Relaton::Index` bsearches on (`Type#candidates_by_number`,
111
+ # `FileIO#deserialize_pubid`). A document family (base, parts,
112
+ # amendments) shares one root and lands in one shard. The key is NOT
113
+ # given a rendered-id fallback: that would break the identity, so a
114
+ # client computing the key from its parsed query would look in a bucket
115
+ # the row is not in and read the miss as not-found. An identifier with
116
+ # no root number keys on "" and lands in shard 0.
117
+ class MachineIndex
118
+ TARGET_ROWS = 15
119
+ MIN_ROWS = 2000
120
+ MIN_SHARDS = 16
121
+ MAX_SHARDS = 65_536
122
+
123
+ # Deliberately does NOT retain the pubid object: `add` extracts the only
124
+ # two things the write path needs (`key`, `id_hash`) while the document
125
+ # streams past, then lets the identifier go. Retaining it costs 3.14 KB
126
+ # per row against 0.44 KB for the derived values alone — 543 MB vs 76 MB
127
+ # on a 177k-row corpus. Anything else derived from pubid must therefore
128
+ # be computed in `add` too; by write time the object is gone.
129
+ Row = Struct.new(:rendered, :file, :key, :id_hash)
130
+
131
+ attr_reader :rows
132
+
133
+ # @param pubid_class [Class] the flavor's pubid Identifier
134
+ # @param index_name [String] the monolith's base name, from the
135
+ # flavor's `INDEXFILE`
136
+ # @param committed [Hash, nil] `{ file => id hash }` read from the
137
+ # repo's own committed index, consulted only when a rendered docid
138
+ # does not parse
139
+ def initialize(pubid_class:, index_name:, committed: nil)
140
+ @pubid_class = pubid_class
141
+ @index_name = index_name
142
+ @committed = committed || {}
143
+ @rows = []
144
+ end
145
+
146
+ def add(rendered, file)
147
+ row = Row.new(rendered, file)
148
+ # Precompute the expensive derived values (root.number walk,
149
+ # lutaml to_hash) once during the streaming pass — computing
150
+ # them at monolith-write time is pathological on 177k rows
151
+ # because each involves object-graph traversal.
152
+ pubid = parse(rendered)
153
+ # A rendered docid the parser rejects falls back to the repo's own
154
+ # committed row: the crawler resolved that id from source metadata.
155
+ # The hash is kept verbatim, so the site's row is byte-identical to
156
+ # the repo's. A row with neither keeps a nil id_hash, which
157
+ # `indexed_rows` drops and `skipped_count` reports.
158
+ hash = pubid ? pubid.to_hash : @committed[file]
159
+ pubid ||= from_hash(hash)
160
+ if pubid
161
+ row.id_hash = hash
162
+ row.key = key_string(pubid)
163
+ end
164
+ @rows << row
165
+ end
166
+
167
+ def count
168
+ @rows.size
169
+ end
170
+
171
+ def key_strategy
172
+ "root-number"
173
+ end
174
+
175
+ # Sized from the rows actually written, the same figure the manifest
176
+ # reports as `count` — not from every row scanned.
177
+ def shard_count
178
+ written = indexed_rows.size
179
+ return 0 if written < MIN_ROWS
180
+
181
+ next_pow2((written.to_f / TARGET_ROWS).ceil).clamp(MIN_SHARDS, MAX_SHARDS)
182
+ end
183
+
184
+ def key_of(row)
185
+ row.key
186
+ end
187
+
188
+ def manifest(generated:)
189
+ {
190
+ "version" => 2,
191
+ # The monolith's base name, so a client can fetch
192
+ # "#{index}.zip" without knowing the flavor's INDEXFILE.
193
+ "index" => @index_name,
194
+ # What the index actually contains, not what was scanned: rows
195
+ # whose id could not be resolved at all are dropped.
196
+ "count" => indexed_rows.size,
197
+ "shards" => shard_count,
198
+ "key" => key_strategy,
199
+ "algorithm" => "crc32",
200
+ "generated" => generated,
201
+ }
202
+ end
203
+
204
+ # { "r" => rendered, "file" => path, "id" => pubid hash }. The same
205
+ # "id" the monolith carries for that row — the two artifacts must not
206
+ # describe one document differently.
207
+ def row_record(row)
208
+ { "r" => row.rendered, "file" => row.file, "id" => row.id_hash }
209
+ end
210
+
211
+ # The rows actually written, in deterministic (key, rendered) order.
212
+ #
213
+ # Carries only rows whose id resolved: the consumer
214
+ # (`Relaton::Index::FileIO#deserialize_id`) calls `from_hash` on every
215
+ # row and raises `InvalidIndexError` on the first one it cannot
216
+ # deserialize, which rejects the *whole* index — so a single unresolved
217
+ # row written as a plain string would poison the file. Five shipping
218
+ # corpora carry a handful of ids that do not parse back from the
219
+ # rendered string (ieee 69, itu-r 47, iec 42, itu 3, nist 3), so this is
220
+ # reachable, not theoretical; `add` rescues those from the committed
221
+ # index, and only what that misses is dropped.
222
+ # `Relaton::Ieee::DataFetcher#build_index` already skips the same rows
223
+ # for the same reason and logs the loss; this matches it.
224
+ #
225
+ # Memoized, and the single place the sort happens — `each_shard` and
226
+ # `write_monolith` previously sorted the corpus independently, so both
227
+ # ran on every build.
228
+ def indexed_rows
229
+ @indexed_rows ||= @rows.reject { |row| row.id_hash.nil? }
230
+ .sort_by { |row| [row.key, row.rendered] }
231
+ end
232
+
233
+ # Rows dropped by `indexed_rows` — reported so the loss is never silent.
234
+ def skipped_count
235
+ count - indexed_rows.size
236
+ end
237
+
238
+ # Rows bucketed by shard, in deterministic (key, rendered) order;
239
+ # empty buckets are absent — a client treats a 404 as not-found.
240
+ def each_shard
241
+ n = shard_count
242
+ return enum_for(:each_shard) unless block_given? && n.positive?
243
+
244
+ buckets = Array.new(n) { [] }
245
+ indexed_rows.each { |row| buckets[Zlib.crc32(row.key) % n] << row_record(row) }
246
+ buckets.each_with_index { |rows, i| yield(format("%05d", i), rows) unless rows.empty? }
247
+ end
248
+
249
+ def monolith_filename
250
+ "#{@index_name}.yaml"
251
+ end
252
+
253
+ # Same shape Relaton::Index::FileIO#save emits: an Array of
254
+ # {id:, file:} hashes, id always a pubid to_hash. Streamed one row at a
255
+ # time — building the full array and calling to_yaml is
256
+ # pathological on six-figure corpora (Psych re-allocates on
257
+ # every nested hash, and the whole array sits in memory).
258
+ # Hand-rolls the YAML instead of calling to_yaml per row:
259
+ # Psych's per-invocation overhead on 177k rows takes ~40 min.
260
+ # The shapes are flat (string id) or one-level-nested (pubid
261
+ # to_hash), so hand-rendering is straightforward and the output
262
+ # is indistinguishable from what Psych produces.
263
+ def write_monolith(path)
264
+ File.open(path, "w:utf-8") do |f|
265
+ f << "---\n"
266
+ indexed_rows.each do |row|
267
+ f << "- :id:\n"
268
+ yaml_nested(f, row.id_hash, " ")
269
+ f << " :file: #{yaml_scalar(row.file)}\n"
270
+ end
271
+ end
272
+ end
273
+
274
+ # An Array is written as a JSON flow sequence, which is valid YAML and
275
+ # escapes every element unambiguously. It must not reach `yaml_scalar`:
276
+ # there `["IEC"].to_s` starts with "[", so it was quoted into the String
277
+ # "[\"IEC\"]" — every ISO/IEC copublished id (`copublishers`) — and
278
+ # `from_hash` cannot cast that, so `FileIO` rejects the whole index.
279
+ def yaml_nested(f, hash, indent)
280
+ hash.each do |k, v|
281
+ case v
282
+ when Hash
283
+ f << "#{indent}#{k}:\n"
284
+ yaml_nested(f, v, indent + " ")
285
+ when Array
286
+ f << "#{indent}#{k}: #{JSON.generate(v)}\n"
287
+ else
288
+ f << "#{indent}#{k}: #{yaml_scalar(v)}\n"
289
+ end
290
+ end
291
+ end
292
+
293
+ # YAML 1.1 plain scalars Psych resolves to true/false/nil. Emitted bare,
294
+ # an id or path with one of these literal values changes *type* on the
295
+ # round trip ("yes" -> true, "null" -> nil).
296
+ YAML11_PLAIN = /\A(?:y|Y|yes|Yes|YES|n|N|no|No|NO|true|True|TRUE|
297
+ false|False|FALSE|on|On|ON|off|Off|OFF|
298
+ null|Null|NULL|~)\z/x
299
+
300
+ def yaml_scalar(value)
301
+ # Booleans and numbers are emitted bare, so they keep their type.
302
+ # Quoting them is what the String branch below exists to prevent,
303
+ # inverted: a pubid `to_hash` carrying a real `true` (CIE's
304
+ # `d_prefix`, 31 of its 1139 rows) came back as the String "true", and
305
+ # the published row no longer matched the repo's own index.
306
+ return "" if value.nil?
307
+ return value.to_s if value == true || value == false || value.is_a?(Integer)
308
+
309
+ s = value.to_s
310
+ # Quote if the value could be ambiguous: leading special char, an
311
+ # embedded ": " or " #", numeric-looking, *trailing* whitespace (the
312
+ # guard here used to read /\z\s/, which can never match — nothing
313
+ # follows end-of-string), a YAML 1.1 plain word, or a C0 control
314
+ # character. The control case is the severe one: emitted raw it makes
315
+ # the whole file unparseable, and one bad row rejects the entire index.
316
+ if s.empty? ||
317
+ s.match?(/\A[-?:,\[\]{}#&*!|>'"%@`\s]|:\s|\s#|\A\d|\s\z/) ||
318
+ s.match?(YAML11_PLAIN) || s.match?(/[[:cntrl:]]/)
319
+ s.inspect
320
+ else
321
+ s
322
+ end
323
+ end
324
+
325
+
326
+
327
+ private
328
+
329
+ def next_pow2(value)
330
+ bit = 0
331
+ bit += 1 while (1 << bit) < value
332
+ 1 << bit
333
+ end
334
+
335
+ def parse(rendered)
336
+ @pubid_class.parse(rendered)
337
+ rescue StandardError
338
+ nil
339
+ end
340
+
341
+ # Only a structured hash: a legacy index under the same name carries
342
+ # plain strings, and one written into a row would break the monolith
343
+ # (`yaml_nested` walks a Hash) and the consumer alike.
344
+ def from_hash(hash)
345
+ hash.is_a?(Hash) && @pubid_class.from_hash(hash) || nil
346
+ rescue StandardError
347
+ nil
348
+ end
349
+
350
+ # The narrowing key `Relaton::Index` sorts and bsearches on. An
351
+ # identifier with no root number gives "", so those rows cluster in
352
+ # shard 0 — the same degeneracy the gem's bsearch already has, and the
353
+ # only shape a client can reproduce without knowing our fallbacks.
354
+ def key_string(pubid)
355
+ pubid.root.number.to_s
356
+ end
357
+ end
358
+
359
+ # @param data_dir [String]
360
+ # @param options [Hash] :output :title :description :favicon :base_url
361
+ # :overwrite :lang :generated :static :shard_size :detail_shard_size
362
+ # :detail
363
+ def self.generate(data_dir, options = {})
364
+ new(data_dir, options).generate
365
+ end
366
+
367
+ def initialize(data_dir, options = {})
368
+ @data_dir = data_dir
369
+ @options = options
370
+ @output = options[:output] || "_site"
371
+ @lang = options[:lang] || "en"
372
+ @overwrite = options.fetch(:overwrite, true)
373
+ @base_url = options[:base_url]
374
+ @title = presence(options[:title]) || "Relaton Index"
375
+ @description = presence(options[:description])
376
+ @favicon = presence(options[:favicon])
377
+ @generated = options.fetch(:generated) { Time.now.utc.strftime("%Y-%m-%d") }
378
+ @include_static = options.fetch(:static, true)
379
+ @shard_size = options.fetch(:shard_size, 5000)
380
+ @detail_shard_size = options.fetch(:detail_shard_size, 500)
381
+ @emit_detail = options.fetch(:detail, true)
382
+ @emit_index = options.fetch(:machine_index, true)
383
+ @publish_data = options.fetch(:publish_data, false)
384
+ @pubid_class = pubid_class_for(options[:flavor])
385
+ @index_name = index_name_for(options[:flavor])
386
+ validate!
387
+ end
388
+
389
+ # @return [String] path to the written index.html
390
+ def generate
391
+ # Read the bundle first: a missing frontend build must fail before any
392
+ # output is written, not halfway through a corpus.
393
+ assets = { "css" => FrontendAssets.stylesheet, "iife" => FrontendAssets.iife }
394
+
395
+ FileUtils.mkdir_p(output)
396
+ purge_stale!
397
+ counts = write_shards
398
+ counts[:index_shards] = write_machine_index_manifest(counts[:total]) if emit_index?
399
+ publish_data! if @publish_data
400
+
401
+ index_path = File.join(output, "index.html")
402
+ write_file(index_path, render(assets, counts))
403
+ copy_404_fallback
404
+ Util.info "Indexed #{counts[:total]} document(s) from #{sources_description} " \
405
+ "into #{counts[:shards]} search shard(s), " \
406
+ "#{counts[:detail_shards]} detail shard(s) and " \
407
+ "#{counts[:index_shards]} machine-index shard(s)"
408
+ index_path
409
+ end
410
+
411
+ private
412
+
413
+ attr_reader :data_dir, :options, :output, :lang, :overwrite, :base_url,
414
+ :title, :description, :favicon, :generated, :include_static,
415
+ :shard_size, :detail_shard_size
416
+
417
+ def emit_detail?
418
+ @emit_detail
419
+ end
420
+
421
+ # `flavor` names the flavor ("iso", "iho", ...) whose pubid Identifier
422
+ # parses docids and whose `INDEXFILE` names the published monolith.
423
+ # Resolved from the two gems' own namespaces — the alias table carries
424
+ # only the names that do not follow the capitalize rule.
425
+ def pubid_class_for(flavor)
426
+ return nil unless flavor_key(flavor)
427
+
428
+ name = namespace_names(flavor).first
429
+ ::Pubid.const_get(name).const_get(:Identifier)
430
+ rescue NameError => e
431
+ raise unless probed_constant?(e, name, :Identifier)
432
+
433
+ raise ArgumentError, "unknown pubid flavor: #{flavor}"
434
+ end
435
+
436
+ # The monolith's base name. Derived from the flavor's own `INDEXFILE`
437
+ # so the site publishes the file that flavor's consumer already fetches
438
+ # — the version there encodes the index *structure*, per flavor, and is
439
+ # not a global generation counter. `index_name:` overrides it for a
440
+ # corpus that is not a relaton flavor.
441
+ def index_name_for(flavor)
442
+ override = presence(options[:index_name])
443
+ return override if override
444
+ return nil unless flavor_key(flavor)
445
+
446
+ name = namespace_names(flavor).last
447
+ ::Relaton.const_get(name).const_get(:INDEXFILE)
448
+ rescue NameError => e
449
+ raise unless probed_constant?(e, name, :INDEXFILE)
450
+
451
+ raise ArgumentError,
452
+ "no relaton flavor for `#{flavor}`; pass index_name to name the index"
453
+ end
454
+
455
+ # True only when the NameError is about the constant we looked up. Looking
456
+ # it up runs the flavor's autoload, and a genuine NameError from inside
457
+ # that file must surface as itself — relabelled "unknown flavor", it
458
+ # would send whoever reads a red deploy to --index-name instead of the bug.
459
+ def probed_constant?(error, *names)
460
+ names.map(&:to_s).include?(error.name.to_s)
461
+ end
462
+
463
+ def flavor_key(flavor)
464
+ key = flavor.to_s.strip.downcase
465
+ key unless key.empty?
466
+ end
467
+
468
+ # [pubid namespace, relaton namespace] for a flavor token. They agree for
469
+ # every flavor but 3GPP, whose pubid namespace is Tgpp.
470
+ def namespace_names(flavor)
471
+ key = flavor_key(flavor)
472
+ FLAVOR_NAMESPACES.fetch(key) do
473
+ name = key.split(/[_-]/).map(&:capitalize).join
474
+ [name, name]
475
+ end
476
+ end
477
+
478
+ def emit_index?
479
+ @emit_index
480
+ end
481
+
482
+ # The data folder's parent — the repo root the index rows' `file` paths
483
+ # and the committed index are relative to.
484
+ def repo_root
485
+ @repo_root ||= File.dirname(File.expand_path(data_dir))
486
+ end
487
+
488
+ # `{ file => id hash }` from the repo's own committed index, the
489
+ # authority for a docid the parser cannot read back from its rendered
490
+ # form. Empty when the repo publishes no index (a fresh repo, or one
491
+ # whose index this build is the first to produce).
492
+ def committed_index
493
+ return @committed_index if defined?(@committed_index)
494
+
495
+ path = File.join(repo_root, "#{@index_name}.yaml")
496
+ @committed_index = File.exist?(path) ? read_committed_index(path) : {}
497
+ rescue StandardError => e
498
+ Util.warn "Ignoring #{path}: #{e.message}"
499
+ @committed_index = {}
500
+ end
501
+
502
+ # Keeps only rows whose `:id` is a structured hash. A legacy index under
503
+ # the same name carries plain strings, and one of those written into a
504
+ # row would break both the monolith (`yaml_nested` walks a Hash) and the
505
+ # consumer (`FileIO#deserialize_id` calls `from_hash`).
506
+ def read_committed_index(path)
507
+ rows = YAML.safe_load(File.read(path), permitted_classes: [Symbol])
508
+ Array(rows).each_with_object({}) do |row, acc|
509
+ next unless row.is_a?(Hash) && row[:id].is_a?(Hash)
510
+
511
+ acc[row[:file]] = row[:id]
512
+ end
513
+ end
514
+
515
+ # nil for a nil/blank option value. A caller workflow that forwards an
516
+ # unset input renders it as an empty string (`--favicon ""`), which must
517
+ # mean "not set" — an empty href in <link rel="icon"> resolves to the page
518
+ # itself, and an empty <meta name="description"> is worse than none.
519
+ def presence(value)
520
+ str = value.to_s.strip
521
+ str unless str.empty?
522
+ end
523
+
524
+ def validate!
525
+ if options.key?(:mode)
526
+ raise ArgumentError,
527
+ "The `mode` option was removed: `relaton index` now always emits " \
528
+ "sharded JSON that the page fetches. Drop --mode from the call."
529
+ end
530
+ unless File.directory?(data_dir)
531
+ raise ArgumentError, "Data directory not found: #{data_dir}"
532
+ end
533
+ { shard_size: shard_size, detail_shard_size: detail_shard_size }
534
+ .each do |name, value|
535
+ next if value.is_a?(Integer) && value.positive?
536
+
537
+ raise ArgumentError, "#{name} must be a positive integer (got #{value.inspect})"
538
+ end
539
+
540
+ # Every data repo publishes a pubid index, so a machine index without
541
+ # a parser could only carry plain-string ids — a shape no consumer
542
+ # narrows on and this generator no longer writes.
543
+ return unless emit_index?
544
+
545
+ if @pubid_class.nil?
546
+ raise ArgumentError,
547
+ "--pubid-flavor is required to build a machine index; " \
548
+ "pass --no-machine-index to build the human site only"
549
+ end
550
+ return if @index_name.match?(INDEX_NAME)
551
+
552
+ raise ArgumentError,
553
+ "index name must be a single file name such as index-v2 (got #{@index_name.inspect})"
554
+ end
555
+
556
+ # Human-readable description of the folders scanned, for the info log.
557
+ # Keeps data_dir as given (no absolute-path noise) and only notes when a
558
+ # sibling static/ was folded in.
559
+ def sources_description
560
+ repo_root = self.repo_root
561
+ static_source_dir(repo_root) ? "#{data_dir} (+ #{STATIC_DIRNAME}/)" : data_dir
562
+ end
563
+
564
+ # ---- streaming ----------------------------------------------------------
565
+
566
+ # The machine-index input: rendered primary docid + repo-relative
567
+ # path (clients prefix their own baseurl onto file).
568
+ def machine_record(item)
569
+ [item["id"], item["yaml_path"]]
570
+ end
571
+
572
+ def write_machine_index_manifest(total)
573
+ machine = @machine_index or return 0
574
+
575
+ FileUtils.mkdir_p(File.join(output, "index"))
576
+ machine.each_shard do |number, rows|
577
+ write_file(File.join(output, "index", "shard-#{number}.json"), JSON.generate(rows))
578
+ end
579
+
580
+ monolith = File.join(output, machine.monolith_filename)
581
+ machine.write_monolith(monolith)
582
+ zip_file(monolith)
583
+
584
+ write_file(
585
+ File.join(output, "index", "manifest.json"),
586
+ JSON.pretty_generate(machine.manifest(generated: @generated)),
587
+ )
588
+ if machine.skipped_count.positive?
589
+ Util.warn "Machine index: skipped #{machine.skipped_count} of " \
590
+ "#{machine.count} document(s) whose id could not be parsed " \
591
+ "(a structured index cannot carry them)."
592
+ end
593
+ machine.shard_count
594
+ end
595
+
596
+ # Zip sibling of a monolith — the form `Relaton::Index url:`
597
+ # fetches today. The zip stores the yaml at its bare basename,
598
+ # matching the layout data repos publish in git.
599
+ def zip_file(yaml_path)
600
+ zip_path = yaml_path.sub(/\.yaml\z/, ".zip")
601
+ File.delete(zip_path) if File.exist?(zip_path)
602
+ Zip::File.open(zip_path, create: true) do |archive|
603
+ archive.add(File.basename(yaml_path), yaml_path)
604
+ end
605
+ end
606
+
607
+ # Copy the scanned corpus onto the site so clients can fetch
608
+ # documents (the index rows' `file` paths) from the same origin.
609
+ # Opt-in: for most repos raw.githubusercontent already serves the
610
+ # committed data, and duplicating a large corpus would double the
611
+ # published-site size against the 1 GB cap.
612
+ def publish_data!
613
+ repo_root = self.repo_root
614
+ source_dirs(repo_root).each do |dir|
615
+ Dir.glob(File.join(dir, "**", "*.{yaml,yml}")).sort.each do |src|
616
+ # Relative to the data dir itself, so the copy lives at
617
+ # output/data/<name>.yaml matching the index rows' file paths.
618
+ rel = Pathname.new(File.expand_path(src))
619
+ .relative_path_from(Pathname.new(dir)).to_s
620
+ dest = File.join(output, "data", rel)
621
+ FileUtils.mkdir_p(File.dirname(dest))
622
+ FileUtils.cp(src, dest)
623
+ end
624
+ end
625
+ end
626
+
627
+ # One pass over the corpus, fanning each document out to both shard
628
+ # families. The search/detail families accumulate nothing — peak memory is
629
+ # one shard of each. The machine index is the exception: shard assignment
630
+ # needs the corpus size, which is not knowable until the pass ends, so
631
+ # `MachineIndex` retains one `Row` per document (measured ~0.44 KB/row — 76 MB at 177k
632
+ # rows; see the note on `Row`, which is why it does not hold the pubid).
633
+ def write_shards
634
+ writer = method(:write_file)
635
+ summary = ShardWriter.new(output, SHARD_PATTERN, shard_size, &writer)
636
+ detail = ShardWriter.new(output, DETAIL_PATTERN, detail_shard_size, &writer)
637
+ if emit_index?
638
+ machine = MachineIndex.new(pubid_class: @pubid_class,
639
+ index_name: @index_name,
640
+ committed: committed_index)
641
+ end
642
+ total = 0
643
+
644
+ each_document do |doc|
645
+ total += 1
646
+ summary << compact_record(doc)
647
+ detail << detail_record(doc) if emit_detail?
648
+ machine&.add(*machine_record(doc))
649
+ end
650
+ summary.flush
651
+ detail.flush
652
+ @machine_index = machine
653
+
654
+ { total: total, shards: summary.count, detail_shards: detail.count }
655
+ end
656
+
657
+ # Yield each index item in corpus order. De-dup is **cross-dir only**: an id
658
+ # already indexed from an *earlier* dir (i.e. static/ duplicating a data/
659
+ # doc) is skipped, so data/ wins — but duplicates *within* a single dir are
660
+ # left as-is, preserving the pre-static behavior of the data scan.
661
+ def each_document
662
+ return to_enum(:each_document) unless block_given?
663
+
664
+ repo_root = self.repo_root
665
+ seen = {}
666
+ source_dirs(repo_root).each do |dir|
667
+ dir_ids = {}
668
+ Dir.glob(File.join(dir, "**", "*.{yaml,yml}")).sort.each do |file|
669
+ item = index_file(file, repo_root)
670
+ next unless item
671
+
672
+ id = dedup_key(item)
673
+ if id && seen.key?(id)
674
+ Util.warn "Skipping #{item['yaml']} (duplicate id #{id}); " \
675
+ "already indexed from #{seen[id]}"
676
+ next
677
+ end
678
+ dir_ids[id] ||= item["yaml"] if id
679
+ # The machine index needs the repo-relative path: clients
680
+ # prefix their own baseurl onto row[:file], so an absolutized
681
+ # ref (yaml_ref with --base-url) would double-prefix.
682
+ item["yaml_path"] = relative_path(file, repo_root)
683
+ yield item
684
+ end
685
+ seen.merge!(dir_ids)
686
+ end
687
+ end
688
+
689
+ # The compact summary record the search shards carry.
690
+ def compact_record(item)
691
+ COMPACT_KEYS.each_with_object({}) { |(key, short), acc| acc[short] = item[key] }
692
+ end
693
+
694
+ # The complement: everything the summary record doesn't carry, tagged with
695
+ # the id so the frontend can verify a positional lookup landed on the right
696
+ # record. nil when the document has no detail fields at all — the slot is
697
+ # still written (as null) so position stays aligned with the summary shards.
698
+ def detail_record(item)
699
+ # "yaml_path" is transport for the machine index, not a detail field.
700
+ rest = item.reject { |key, _| COMPACT_KEYS.key?(key) || key == "yaml_path" }
701
+ return nil if rest.empty?
702
+
703
+ { "r" => item["id"] }.merge(rest)
704
+ end
705
+
706
+ # The data folder, plus an auto-detected sibling static/ folder (its bib
707
+ # docs are part of the corpus). Data is scanned first so it wins on a
708
+ # cross-dir duplicate id. Enabled by default; --no-static opts out.
709
+ def source_dirs(repo_root)
710
+ [data_dir, static_source_dir(repo_root)].compact
711
+ end
712
+
713
+ # The sibling static/ dir to fold in, or nil when disabled, absent, or the
714
+ # same folder as data_dir (guards `relaton index static` double-scanning).
715
+ def static_source_dir(repo_root)
716
+ return nil unless include_static
717
+
718
+ static = File.join(repo_root, STATIC_DIRNAME)
719
+ return nil unless File.directory?(static)
720
+ return nil if File.expand_path(static) == File.expand_path(data_dir)
721
+
722
+ static
723
+ end
724
+
725
+ # The id we de-dup on, or nil when the doc has no usable id (a blank/empty
726
+ # id must NOT collapse distinct docid-less documents together).
727
+ def dedup_key(item)
728
+ id = item["id"]
729
+ id unless id.nil? || id.empty?
730
+ end
731
+
732
+ # Normalize one YAML file to an index item, or nil if it's a machine index,
733
+ # not a document, or unparseable.
734
+ def index_file(file, repo_root)
735
+ return if File.basename(file).match?(SKIP_BASENAMES)
736
+
737
+ doc = load_yaml(file)
738
+ return unless document?(doc)
739
+
740
+ rel = relative_path(file, repo_root)
741
+ IndexItemNormalizer.normalize(doc, lang: lang, yaml_ref: yaml_ref(rel))
742
+ rescue Psych::SyntaxError => e
743
+ Util.warn "Skipping #{file}: #{e.message}"
744
+ nil
745
+ end
746
+
747
+ def load_yaml(file)
748
+ content = File.read(file, encoding: "utf-8")
749
+ begin
750
+ YAML.safe_load(content, permitted_classes: [Date, Time], aliases: true)
751
+ rescue ArgumentError
752
+ # older Psych positional signature
753
+ YAML.safe_load(content, [Date, Time], [], true)
754
+ end
755
+ end
756
+
757
+ # A bib document (not a collection/index) — has an id/docid/title, no "root".
758
+ def document?(doc)
759
+ return false unless doc.is_a?(Hash)
760
+ return false if doc.key?("root")
761
+
762
+ doc.key?("id") || doc.key?("docidentifier") || doc.key?("title")
763
+ end
764
+
765
+ def relative_path(file, repo_root)
766
+ Pathname.new(File.expand_path(file))
767
+ .relative_path_from(Pathname.new(repo_root)).to_s
768
+ end
769
+
770
+ def yaml_ref(rel)
771
+ return rel if base_url.nil? || base_url.empty?
772
+
773
+ "#{base_url.chomp('/')}/#{rel}"
774
+ end
775
+
776
+ # ---- rendering ----------------------------------------------------------
777
+
778
+ def render(assets, counts)
779
+ env = Liquid::Environment.build(
780
+ file_system: Liquid::LocalFileSystem.new(TEMPLATE_DIR),
781
+ )
782
+ template = Liquid::Template.parse(
783
+ File.read(File.join(TEMPLATE_DIR, "page.liquid"), encoding: "utf-8"),
784
+ environment: env,
785
+ )
786
+ template.render!(
787
+ "title" => title,
788
+ "description" => description,
789
+ "favicon" => favicon,
790
+ "favicon_type" => favicon_type,
791
+ "css" => assets["css"],
792
+ "iife" => assets["iife"],
793
+ "generated" => generated,
794
+ "total" => counts[:total],
795
+ "shards" => counts[:shards],
796
+ "shard_size" => shard_size,
797
+ "detail_shards" => counts[:detail_shards],
798
+ "detail_shard_size" => detail_shard_size,
799
+ )
800
+ end
801
+
802
+ # The <link rel="icon"> type for the configured favicon, or nil when there
803
+ # is none or its extension isn't a known image type. The href is passed
804
+ # through verbatim (absolute URL or output-relative path alike), so any
805
+ # ?query/#fragment is stripped before looking at the extension.
806
+ def favicon_type
807
+ return nil unless favicon
808
+
809
+ FAVICON_TYPES[File.extname(favicon.split(/[?#]/, 2).first.to_s).downcase]
810
+ end
811
+
812
+ # Drop shards from a previous build. Without a manifest, an orphan shard
813
+ # left by a larger corpus is invisible — nothing points at it — so it would
814
+ # sit in the deployed site indefinitely.
815
+ # Path-based document URLs (/doc/<id>) are served by this fallback:
816
+ # GitHub Pages has no rewrites, so unknown paths resolve to 404.html,
817
+ # whose script forwards the id into the app as ?doc=<id>.
818
+ def copy_404_fallback
819
+ fallback = File.join(FrontendAssets.dist_dir, "404.html")
820
+ return unless File.exist?(fallback)
821
+
822
+ write_file(File.join(output, "404.html"), File.read(fallback))
823
+ end
824
+
825
+ def purge_stale!
826
+ return unless overwrite
827
+
828
+ # Read before the glob deletes it. `--index-name` accepts a name
829
+ # `index-v*` does not match, so the previous build's own manifest is
830
+ # the only record of which monolith it wrote.
831
+ previous = previous_index_name
832
+ # `base:` rather than interpolating `output` into the pattern: an output
833
+ # path containing glob metacharacters (`[`, `{`, `*`, `?`, …) would
834
+ # otherwise match nothing, and the stale shards would survive silently.
835
+ paths = Dir.glob(STALE_GLOBS, base: output).map { |name| File.join(output, name) }
836
+ paths += %w[yaml zip].map { |ext| File.join(output, "#{previous}.#{ext}") } if previous
837
+ paths.uniq.each { |path| File.delete(path) if File.file?(path) }
838
+ end
839
+
840
+ # The monolith name a previous build recorded, if it is a safe basename.
841
+ def previous_index_name
842
+ path = File.join(output, "index", "manifest.json")
843
+ return unless File.file?(path)
844
+
845
+ name = JSON.parse(File.read(path))["index"]
846
+ name if name.is_a?(String) && name.match?(INDEX_NAME)
847
+ rescue JSON::ParserError
848
+ nil
849
+ end
850
+
851
+ def write_file(path, content)
852
+ if File.exist?(path) && !overwrite
853
+ Util.warn "Skipping existing #{path} (use --overwrite)"
854
+ return
855
+ end
856
+ File.write(path, content, encoding: "utf-8")
857
+ end
858
+ end
859
+ end
860
+ end