metanorma 2.0.0 → 2.5.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (68) hide show
  1. checksums.yaml +4 -4
  2. data/.envrc +1 -0
  3. data/.gitignore +3 -0
  4. data/.rubocop.yml +23 -2
  5. data/Gemfile +17 -0
  6. data/Gemfile.devel +10 -0
  7. data/README.adoc +1 -1
  8. data/lib/metanorma/collection/artifact_store.rb +142 -0
  9. data/lib/metanorma/collection/collection.rb +160 -122
  10. data/lib/metanorma/collection/config/bibdata.rb +3 -3
  11. data/lib/metanorma/collection/config/compile_options.rb +4 -5
  12. data/lib/metanorma/collection/config/config.rb +81 -83
  13. data/lib/metanorma/collection/config/converters.rb +92 -8
  14. data/lib/metanorma/collection/config/directive.rb +9 -3
  15. data/lib/metanorma/collection/config/doc_container.rb +28 -0
  16. data/lib/metanorma/collection/config/manifest.rb +59 -29
  17. data/lib/metanorma/collection/config/namespaces.rb +12 -0
  18. data/lib/metanorma/collection/document/document.rb +56 -15
  19. data/lib/metanorma/collection/filelookup/filelookup.rb +185 -77
  20. data/lib/metanorma/collection/filelookup/filelookup_sectionsplit.rb +99 -27
  21. data/lib/metanorma/collection/filelookup/utils.rb +52 -0
  22. data/lib/metanorma/collection/helpers.rb +82 -0
  23. data/lib/metanorma/collection/log.rb +24 -0
  24. data/lib/metanorma/collection/manifest/manifest.rb +45 -14
  25. data/lib/metanorma/collection/multilingual/multilingual.rb +660 -0
  26. data/lib/metanorma/collection/renderer/filelocation.rb +166 -0
  27. data/lib/metanorma/collection/renderer/fileparse.rb +135 -115
  28. data/lib/metanorma/collection/renderer/fileprocess.rb +185 -37
  29. data/lib/metanorma/collection/renderer/navigation.rb +21 -7
  30. data/lib/metanorma/collection/renderer/render_word.rb +15 -10
  31. data/lib/metanorma/collection/renderer/renderer.rb +236 -32
  32. data/lib/metanorma/collection/renderer/svg.rb +107 -0
  33. data/lib/metanorma/collection/renderer/utils.rb +110 -39
  34. data/lib/metanorma/collection/sectionsplit/collection.rb +110 -0
  35. data/lib/metanorma/collection/sectionsplit/sectionsplit.rb +198 -115
  36. data/lib/metanorma/collection/util/disambig_files.rb +4 -5
  37. data/lib/metanorma/collection/util/util.rb +155 -18
  38. data/lib/metanorma/collection/xrefprocess/xrefprocess.rb +47 -33
  39. data/lib/metanorma/compile/assets/icc-boilerplate.adoc +41 -0
  40. data/lib/metanorma/compile/compile.rb +181 -156
  41. data/lib/metanorma/compile/compile_options.rb +186 -79
  42. data/lib/metanorma/compile/extract.rb +75 -58
  43. data/lib/metanorma/compile/flavor.rb +28 -0
  44. data/lib/metanorma/compile/output_filename.rb +75 -0
  45. data/lib/metanorma/compile/output_filename_config.rb +27 -0
  46. data/lib/metanorma/compile/relaton_drop.rb +56 -0
  47. data/lib/metanorma/compile/render.rb +226 -0
  48. data/lib/metanorma/compile/validator.rb +26 -0
  49. data/lib/metanorma/compile/writeable.rb +12 -0
  50. data/lib/metanorma/util/fontist_helper.rb +35 -19
  51. data/lib/metanorma/version.rb +1 -1
  52. data/lib/metanorma.rb +1 -7
  53. data/metanorma.gemspec +15 -21
  54. data/plans/collection-rendering-architecture-review.md +516 -0
  55. metadata +65 -113
  56. data/.hound.yml +0 -5
  57. data/lib/metanorma/asciidoctor_extensions/glob_include_processor.rb +0 -22
  58. data/lib/metanorma/asciidoctor_extensions.rb +0 -1
  59. data/lib/metanorma/compile/compile_validate.rb +0 -68
  60. data/lib/metanorma/config/config.rb +0 -23
  61. data/lib/metanorma/input/asciidoc.rb +0 -107
  62. data/lib/metanorma/input/base.rb +0 -9
  63. data/lib/metanorma/input.rb +0 -7
  64. data/lib/metanorma/processor/processor.rb +0 -54
  65. data/lib/metanorma/processor.rb +0 -6
  66. data/lib/metanorma/registry/registry.rb +0 -63
  67. data/lib/metanorma/shale_monkeypatch.rb +0 -15
  68. data/lib/metanorma/util/util.rb +0 -45
@@ -1,7 +1,9 @@
1
1
  require "isodoc"
2
2
  require "htmlentities"
3
3
  require "metanorma-utils"
4
+ require "marcel"
4
5
  require_relative "filelookup_sectionsplit"
6
+ require_relative "utils"
5
7
 
6
8
  module Metanorma
7
9
  class Collection
@@ -18,38 +20,54 @@ module Metanorma
18
20
  @parent = parent
19
21
  @xml = parent.xml
20
22
  @isodoc = parent.isodoc
23
+ @isodoc_presxml = parent.isodoc_presxml
21
24
  @path = path
22
25
  @compile = parent.compile
23
26
  @documents = parent.documents
24
27
  @files_to_delete = []
25
28
  @disambig = Util::DisambigFiles.new
26
29
  @manifest = parent.manifest
27
- read_files(@manifest.entry)
30
+ read_files(@manifest.entry, parent.manifest)
28
31
  end
29
32
 
30
- def read_files(entries)
33
+ def read_files(entries, parent, idx = 0)
31
34
  Array(entries).each do |e|
32
- e.file and read_file(e)
33
- read_files(e.entry)
35
+ derive_format(e, parent)
36
+ if e.file
37
+ read_file(e, idx)
38
+ idx += 1
39
+ end
40
+ idx = read_files(e.entry, e, idx)
34
41
  end
42
+ idx
35
43
  end
36
44
 
37
- def read_file(manifest)
45
+ def derive_format(entry, parent)
46
+ entry.attachment and return
47
+ if Array(entry.format).empty?
48
+ parent_fmt = Array(parent.format)
49
+ entry.format = parent_fmt.empty? ? %w(xml presentation html) : parent_fmt.dup
50
+ end
51
+ entry.format |= ["xml", "presentation"]
52
+ end
53
+
54
+ def read_file(manifest, idx)
38
55
  i, k = read_file_idents(manifest)
39
- entry = file_entry(manifest, k) or return
56
+ entry = file_entry(manifest, k, idx) or return
40
57
  bibdata_process(entry, i)
41
58
  bibitem_process(entry)
42
- @files[key(i)] = entry
59
+ @files[entry_key(i)] = entry
43
60
  end
44
61
 
45
62
  def read_file_idents(manifest)
46
63
  id = manifest.identifier
47
- sanitised_id = key(@isodoc.docid_prefix("", manifest.identifier.dup))
64
+ sanitised_id =
65
+ entry_key(@isodoc.docid_prefix("", manifest.identifier.dup))
48
66
  # if manifest.bibdata and # NO, DO NOT FISH FOR THE GENUINE IDENTIFIER IN BIBDATA
49
67
  # d = manifest.bibdata.docidentifier.detect { |x| x.primary } ||
50
68
  # manifest.bibdata.docidentifier.first
51
69
  # k = d.id
52
- # i = key(@isodoc.docid_prefix(d.type, d.id.dup))
70
+ # i = entry_key(@isodoc.docid_prefix(d.type, d.id.dup))
53
71
  # end
54
72
  [id, sanitised_id]
55
73
  end
@@ -62,12 +80,18 @@ module Metanorma
62
80
  file, _filename = targetfile(entry, read: true)
63
81
  xml = Nokogiri::XML(file, &:huge)
64
82
  add_document_suffix(ident, xml)
65
- entry.merge!(anchors: read_anchors(xml), ids: read_ids(xml),
66
- bibdata: xml.at(ns("//bibdata")),
67
- document_suffix: xml.root["document_suffix"])
83
+ entry.merge!(bibdata_extract(xml))
68
84
  end
69
85
  end
70
86
 
87
+ def bibdata_extract(xml)
88
+ anchors = read_anchors(xml)
89
+ { anchors: anchors, anchors_lookup: anchors_lookup(anchors),
90
+ ids: read_ids(xml),
91
+ bibdata: xml.at(ns("//bibdata")),
92
+ document_suffix: xml.root["document_suffix"] }
93
+ end
94
+
71
95
  def bibitem_process(entry)
72
96
  entry[:bibitem] = entry[:bibdata].dup
73
97
  entry[:bibitem].name = "bibitem"
@@ -76,57 +100,138 @@ module Metanorma
76
100
  end
77
101
 
78
102
  # ref is the absolute source file address
79
- # rel_path is the relative source file address, relative to the YAML locaton
103
+ # rel_path is the relative source file address, relative to the YAML location
80
104
  # out_path is the destination file address, with any references outside
81
105
  # the working directory (../../...) truncated, and based on relative path
82
106
  # identifier is the id with only spaces, no nbsp
83
- def file_entry(ref, identifier)
107
+ # idx is the index of the document in the manifest
108
+ def file_entry(ref, identifier, idx)
84
109
  ref.file or return
85
110
  abs = @documents[Util::key identifier].file
111
+ # For sectionsplit outputs from YAML manifest, we need to compute the full path
112
+ # by combining sectionsplit_filename directory with ref.file basename
113
+ sso = ref.respond_to?(:sectionsplit_output) && ref.sectionsplit_output
114
+ out_path, rel_path = file_entry_paths(ref, idx, sso)
86
115
  ret = if ref.file
87
- { type: "fileref", ref: abs, rel_path: ref.file, url: ref.url,
88
- out_path: output_file_path(ref) }
116
+ { type: "fileref", ref: abs, rel_path: rel_path, url: ref.url,
117
+ out_path: out_path, idx: idx,
118
+ output_filename: ref.output_filename,
119
+ sectionsplit_filename: ref.sectionsplit_filename,
120
+ pdffile: ref.pdffile, format: ref.format&.map(&:to_sym) }
121
+ .compact
89
122
  else { type: "id", ref: ref.id }
90
123
  end
91
124
  file_entry_copy(ref, ret)
92
125
  ret.compact
93
126
  end
94
127
 
128
+
129
+ # ref is the absolute source file address
130
+ # rel_path is the relative source file address, relative to the YAML location
131
+ # out_path is the destination file address, with any references outside
132
+ # the working directory (../../...) truncated, and based on relative path
133
+ # identifier is the id with only spaces, no nbsp
134
+ # extract_opts are the compilation options extracted as document attributes
135
+ def file_entry_struct(ref, abs)
136
+ adoc = abs.sub(/\.xml$/, ".adoc")
137
+ if adoc.end_with?(".adoc") && File.exist?(adoc)
138
+ opts = Metanorma::Input::Asciidoc.new.extract_options(File.read(adoc))
139
+ end
140
+ { type: "fileref", ref: abs, rel_path: ref.file, url: ref.url,
141
+ out_path: output_file_path(ref), pdffile: ref.pdffile,
142
+ format: ref.format&.map(&:to_sym), extract_opts: opts }.compact
143
+ end
144
+
145
+ def file_entry_paths(ref, idx, sso)
146
+ base = File.basename(ref.file, ".xml")
147
+ if sso && ref.respond_to?(:sectionsplit_filename) &&
148
+ ref.sectionsplit_filename
149
+ # Extract directory from sectionsplit_filename
150
+ dir = File.dirname(ref.sectionsplit_filename)
151
+ if dir == "." # No directory in pattern
152
+ [output_file_path(ref, idx), ref.file]
153
+ else # Pattern has directory, prepend it
154
+ full_path = File.join(dir, base)
155
+ [full_path, "#{full_path}.xml"]
156
+ end
157
+ else [output_file_path(ref, idx), ref.file]
158
+ end
159
+ end
160
+
161
+ # Substitute special strings in filename patterns.
162
+ # See Util.substitute_filename_pattern.
163
+ def substitute_filename_pattern(pattern, options = {})
164
+ Util.substitute_filename_pattern(pattern, options)
165
+ end
166
+
95
167
  # TODO make the output file location reflect source location universally,
96
168
  # not just for attachments: no File.basename
97
- def output_file_path(ref)
98
- f = File.basename(ref.file)
99
- ref.attachment and f = ref.file
100
- @disambig.source2dest_filename(f)
169
+ #
170
+ # For files with custom directory structure, construct path with directory
171
+ # For files with output_filename, use that (with substitutions)
172
+ # For others, use basename of ref.file
173
+ def output_file_path(ref, idx)
174
+ has_custom_dir, file_has_dir, params = output_file_path_prep(ref, idx)
175
+ # Apply sectionsplit_filename directory structure if:
176
+ # 1. File has sectionsplit enabled (parent document being split), OR
177
+ # 2. File is a sectionsplit output (from collection or single-file sectionsplit)
178
+ # Regular files that inherit sectionsplit_filename from collection level
179
+ # but are not sectionsplit outputs should NOT use it
180
+ is_sectionsplit_output = ref.respond_to?(:sectionsplit_output) && ref.sectionsplit_output
181
+ use_sectionsplit_dir = ref.sectionsplit_filename && has_custom_dir &&
182
+ (ref.sectionsplit || is_sectionsplit_output || file_has_dir)
183
+ f = if use_sectionsplit_dir
184
+ # For sectionsplit outputs, return just the basename
185
+ # The directory will be applied during file_compile_format
186
+ # via preserve_directory_structure?
187
+ File.basename(ref.file)
188
+ elsif ref.output_filename
189
+ substitute_filename_pattern(ref.output_filename, **params)
190
+ elsif file_has_dir
191
+ ref.file # Preserve directory structure already in ref.file
192
+ elsif ref.attachment
193
+ ref.file
194
+ else File.basename(ref.file)
195
+ end
196
+ ret = @disambig.source2dest_filename(f, preserve_dirs: ref.attachment)
197
+ warn ret
198
+ ret
199
+ end
200
+
201
+ def output_file_path_prep(ref, idx)
202
+ b = File.basename(ref.file)
203
+ b_no_ext = File.basename(ref.file, ".*")
204
+ # Check for sectionsplit_filename (for both parent and split output files)
205
+ # or output_filename
206
+ custom_filename = ref.sectionsplit_filename || ref.output_filename
207
+ has_custom_dir = custom_filename && File.dirname(custom_filename) != "."
208
+ # Also check if ref.file itself contains a directory
209
+ file_has_dir = File.dirname(ref.file) != "."
210
+ params = { document_num: idx, basename: b_no_ext, basename_legacy: b }
211
+ [has_custom_dir, file_has_dir, params]
101
212
  end
102
213
 
103
214
  def file_entry_copy(ref, ret)
104
215
  %w(attachment sectionsplit index presentation-xml url
105
- bare-after-first).each do |s|
216
+ bare-after-first output_filename sectionsplit_filename
217
+ sectionsplit_output).each do |s|
106
218
  ref.respond_to?(s.to_sym) and
107
- ret[s.gsub("-", "").to_sym] = ref.send(s)
219
+ ret[s.delete("-").to_sym] = ref.send(s)
108
220
  end
109
221
  end
110
222
 
111
223
  def add_document_suffix(identifier, doc)
112
224
  document_suffix = Metanorma::Utils::to_ncname(identifier)
113
- Metanorma::Utils::anchor_attributes.each do |(tag_name, attribute_name)|
114
- Util::add_suffix_to_attributes(doc, document_suffix, tag_name,
115
- attribute_name, @isodoc)
225
+ ids = doc.xpath("./@id | .//@id").map(&:value)
226
+ Util::anchor_id_attributes.each do |(tag_name, attr_name)|
227
+ Util::add_suffix_to_attrs(doc, document_suffix, tag_name, attr_name,
228
+ @isodoc)
116
229
  end
117
- url_in_css_styles(doc, document_suffix)
230
+ Util::url_in_css_styles(doc, ids, document_suffix)
118
231
  doc.root["document_suffix"] ||= ""
119
232
  doc.root["document_suffix"] += document_suffix
120
233
  end
121
234
 
122
- # update relative URLs, url(#...), in CSS in @style attrs (including SVG)
123
- def url_in_css_styles(doc, document_suffix)
124
- doc.xpath("//*[@style]").each do |s|
125
- s["style"] = s["style"]
126
- .gsub(%r{url\(#([^)]+)\)}, "url(#\\1_#{document_suffix})")
127
- end
128
- end
129
-
130
235
  # return citation url for file
131
236
  # @param doc [Boolean] I am a Metanorma document,
132
237
  # so my URL should end with html or pdf or whatever
@@ -135,13 +240,6 @@ module Metanorma
135
240
  data[:url] || targetfile(data, options)[1]
136
241
  end
137
242
 
138
- # are references to the file to be linked to a file in the collection,
139
- # or externally? Determines whether file suffix anchors are to be used
140
- def url?(ident)
141
- data = get(ident) or return false
142
- data[:url]
143
- end
144
-
145
243
  # return file contents + output filename for each file in the collection,
146
244
  # given a docref entry
147
245
  # @param data [Hash] docref entry
@@ -155,7 +253,7 @@ module Metanorma
155
253
  options = { read: false, doc: true, relative: false }.merge(options)
156
254
  path = options[:relative] ? data[:rel_path] : data[:ref]
157
255
  if data[:type] == "fileref"
158
- ref_file path, data[:out_path], options[:read], options[:doc]
256
+ ref_file path, data, options[:read], options[:doc]
159
257
  else
160
258
  xml_file data[:id], options[:read]
161
259
  end
@@ -165,13 +263,40 @@ module Metanorma
165
263
  targetfile(get(ident), options)
166
264
  end
167
265
 
168
- def ref_file(ref, out, read, doc)
266
+ def ref_file(ref, data, read, doc)
169
267
  file = File.read(ref, encoding: "utf-8") if read
170
- filename = out.dup
171
- filename.sub!(/\.xml$/, ".html") if doc
268
+ # Use the actual output path from :outputs if available (set after compilation)
269
+ # Otherwise fall back to :out_path (set at initialization)
270
+ filename = if doc && data[:outputs] && data[:outputs][:html]
271
+ data[:outputs][:html].sub(
272
+ %r{^#{Regexp.escape(@parent.outdir)}/}, ""
273
+ )
274
+ else
275
+ data[:out_path].dup
276
+ end
277
+ if doc && !data[:outputs]
278
+ filename = ref_file_xml2html(filename)
279
+ end
172
280
  [file, filename]
173
281
  end
174
282
 
283
+ # Check if file has a recognized MIME type (other than XML)
284
+ # If so, don't append .html (e.g., .svg, .png, .jpg, etc.)
285
+ # Only process if it doesn't have a recognized non-XML extension
286
+ # If filename ends in .xml, replace with .html
287
+ # Otherwise (including sectionsplit files like "file.xml.0" or
288
+ # custom titles), append .html
289
+ def ref_file_xml2html(filename)
290
+ unless Util::mime_file_recognised?(filename) &&
291
+ !filename.end_with?(".xml")
292
+ filename = if filename.end_with?(".xml")
293
+ filename.sub(/\.xml$/, ".html")
294
+ else "#{filename}.html"
295
+ end
296
+ end
297
+ filename
298
+ end
299
+
175
300
  def xml_file(id, read)
176
301
  file = @xml.at(ns("//doc-container[@id = '#{id}']")).to_xml if read
177
302
  filename = "#{id}.html"
@@ -198,10 +323,10 @@ module Metanorma
198
323
  ret[val[:type]] ||= {}
199
324
  index = if val[:container] || val[:label].nil? || val[:label].empty?
200
325
  UUIDTools::UUID.random_create.to_s
201
- else val[:label]
326
+ else val[:label].gsub(%r{<[^>]+>}, "")
202
327
  end
203
328
  ret[val[:type]][index] = key
204
- ret[val[:type]][val[:value]] = key if val[:value]
329
+ v = val[:value] and ret[val[:type]][v.gsub(%r{<[^>]+>}, "")] = key
205
330
  end
206
331
 
207
332
  # Also parse all ids in doc (including ones which won't be xref targets)
@@ -209,41 +334,24 @@ module Metanorma
209
334
  ret = {}
210
335
  xml.traverse do |x|
211
336
  x.text? and next
212
- /^semantic__/.match?(x.name) and next
213
337
  x["id"] and ret[x["id"]] = true
214
338
  end
215
339
  ret
216
340
  end
217
341
 
218
- def key(ident)
219
- @c.decode(ident).gsub(/(\p{Zs})+/, " ").sub(/^metanorma-collection /,
220
- "")
221
- end
222
-
223
- def keys
224
- @files.keys
225
- end
226
-
227
- def get(ident, attr = nil)
228
- if attr then @files[key(ident)][attr]
229
- else @files[key(ident)]
230
- end
231
- end
232
-
233
- def set(ident, attr, value)
234
- @files[key(ident)][attr] = value
235
- end
236
-
237
- def each
238
- @files.each
239
- end
240
-
241
- def each_with_index
242
- @files.each_with_index
243
- end
244
-
245
- def ns(xpath)
246
- @isodoc.ns(xpath)
342
+ # Check if we should preserve directory structure for an identifier
343
+ # Returns the custom filename if directory structure should be preserved,
344
+ # nil otherwise
345
+ def preserve_directory_structure?(ident)
346
+ ret = if get(ident, :sectionsplit_output)
347
+ # For sectionsplit outputs, use rel_path which has the directory
348
+ get(ident, :rel_path) || get(ident, :out_path)
349
+ elsif get(ident, :sectionsplit)
350
+ get(ident, :sectionsplit_filename)
351
+ else get(ident, :output_filename)
352
+ end
353
+ # Return the custom filename only if it contains a directory
354
+ ret && File.dirname(ret) != "." ? ret : nil
247
355
  end
248
356
  end
249
357
  end
@@ -6,8 +6,8 @@ module Metanorma
6
6
  def add_section_split
7
7
  ret = @files.keys.each_with_object({}) do |k, m|
8
8
  if @files[k][:sectionsplit] && !@files[k][:attachment]
9
- process_section_split_instance(k, m)
10
- cleanup_section_split_instance(k, m)
9
+ original_out_path = process_section_split_instance(k, m)
10
+ cleanup_section_split_instance(k, m, original_out_path)
11
11
  end
12
12
  m[k] = @files[k]
13
13
  end
@@ -15,67 +15,139 @@ module Metanorma
15
15
  end
16
16
 
17
17
  def process_section_split_instance(key, manifest)
18
+ # Save the original out_path before it gets modified
19
+ original_out_path = @files[key][:out_path]
18
20
  s, sectionsplit_manifest = sectionsplit(key)
19
21
  s.each_with_index do |f1, i|
20
22
  add_section_split_instance(f1, manifest, key, i)
21
23
  end
22
- manifest["#{key}:index.html"] =
23
- add_section_split_cover(sectionsplit_manifest, key)
24
+ a = add_section_split_attachments(sectionsplit_manifest, key) and
25
+ manifest["#{key}:attachments"] = a
26
+ add_section_split_cover(manifest, sectionsplit_manifest, key)
27
+ # Return the original path for cleanup
28
+ original_out_path
24
29
  end
25
30
 
26
- def cleanup_section_split_instance(key, manifest)
31
+ def cleanup_section_split_instance(key, manifest, original_out_path)
32
+ # Delete the sectionsplit index.html from source directory after it's copied to output
27
33
  @files_to_delete << manifest["#{key}:index.html"][:ref]
34
+ # Delete the original files when sectionsplit happens (all formats: html, xml, presentation.xml)
35
+ # Use the saved original out_path (before it was changed to index.html)
36
+ if original_out_path
37
+ base = File.join(@parent.outdir, original_out_path.sub(/\.xml$/, ""))
38
+ @files_to_delete << "#{base}.html"
39
+ @files_to_delete << "#{base}.xml"
40
+ @files_to_delete << "#{base}.presentation.xml"
41
+ end
28
42
  # @files[key].delete(:ids).delete(:anchors)
29
43
  @files[key][:indirect_key] = @sectionsplit.key
44
+ # sectionsplit is HTML-only; hand the parent its whole Presentation XML
45
+ # so the collection PDF/presentation concatenation inlines the document
46
+ # intact (raw_file) instead of degrading to a bibdata-only, namespace-
47
+ # less stub. file_compile early-returns for sectionsplit, so this
48
+ # :outputs entry is what concatenate1 consumes.
49
+ # https://github.com/metanorma/iso-10303/issues/208
50
+ @files[key][:outputs] =
51
+ { presentation: @sectionsplit.whole_presentation_file }
30
52
  end
31
53
 
32
- def add_section_split_cover(manifest, ident)
54
+ def add_section_split_cover(manifest, sectionsplit_manifest, ident)
33
55
  cover = @sectionsplit
34
- .section_split_cover(manifest, @parent.dir_name_cleanse(ident),
56
+ .section_split_cover(sectionsplit_manifest,
57
+ @parent.dir_name_cleanse(ident),
35
58
  one_doc_collection?)
36
59
  @files[ident][:out_path] = cover
37
- { attachment: true, index: false, out_path: cover,
38
- ref: File.join(File.dirname(manifest.file), cover) }
60
+ src = File.join(File.dirname(sectionsplit_manifest.file), cover)
61
+ m = { attachment: true, index: false, out_path: cover, ref: src }
62
+ manifest["#{ident}:index.html"] = m
63
+ one_doc_collection? and
64
+ add_cover_one_doc_coll(manifest, sectionsplit_manifest, ident, m)
65
+ end
66
+
67
+ def add_cover_one_doc_coll(manifest, sectionsplit_manifest, key, entry)
68
+ idx = File.join(File.dirname(sectionsplit_manifest.file), "index.html")
69
+ FileUtils.cp entry[:ref], idx
70
+ manifest["#{key}:index1.html"] =
71
+ entry.merge(out_path: "index.html", ref: idx)
39
72
  end
40
73
 
41
74
  def one_doc_collection?
42
- return false
43
75
  docs = 0
44
76
  @files.each_value do |v|
45
77
  v[:attachment] and next
46
78
  v[:presentationxml] and next
47
79
  docs += 1
48
80
  end
49
- docs > 1
81
+ docs <= 1
82
+ end
83
+
84
+ def add_section_split_attachments(manifest, ident)
85
+ attachments = @sectionsplit
86
+ .section_split_attachments(out: File.dirname(manifest.file))
87
+ attachments or return
88
+ @files[ident][:out_path] = attachments
89
+ { attachment: true, index: false, out_path: attachments,
90
+ ref: File.join(File.dirname(manifest.file), attachments) }
50
91
  end
51
92
 
52
93
  def add_section_split_instance(file, manifest, key, idx)
53
- presfile, newkey, xml =
54
- add_section_split_instance_prep(file, key)
55
- manifest[newkey] =
56
- { parentid: key, presentationxml: true, type: "fileref",
57
- rel_path: file[:url], out_path: File.basename(file[:url]),
58
- anchors: read_anchors(xml), ids: read_ids(xml),
59
- sectionsplit_output: true,
60
- bibdata: @files[key][:bibdata], ref: presfile }
61
- @files_to_delete << file[:url]
62
- manifest[newkey][:bare] = true unless idx.zero?
94
+ presfile, newkey, xml = add_section_split_instance_prep(file, key)
95
+ anchors = read_anchors(xml)
96
+ # Preserve directory structure in out_path if parent has custom sectionsplit_filename with directory
97
+ sectionsplit_fname = @files[key][:sectionsplit_filename]
98
+
99
+ # file[:url] from sectionsplit.rb already has placeholders substituted and includes full path
100
+ # Use it directly for out_path (without .xml extension)
101
+ base_filename = File.basename(file[:url], ".xml")
102
+
103
+ # Get the directory from file[:url] which already has placeholders substituted
104
+ file_dir = File.dirname(file[:url])
105
+
106
+ # If file[:url] has a directory (i.e., placeholders were substituted), use it
107
+ out_path_value = if file_dir == "."
108
+ base_filename
109
+ else
110
+ File.join(file_dir, base_filename)
111
+ end
112
+
113
+ m = { parentid: key, presentationxml: true, type: "fileref",
114
+ rel_path: out_path_value, out_path: out_path_value,
115
+ anchors: anchors, anchors_lookup: anchors_lookup(anchors),
116
+ ids: read_ids(xml), format: @files[key][:format],
117
+ sectionsplit_output: true, indirect_key: @sectionsplit.key,
118
+ bibdata: @files[key][:bibdata], ref: presfile,
119
+ sectionsplit_filename: sectionsplit_fname,
120
+ idx: @files[key][:idx] }
121
+ m[:bare] = true unless idx.zero?
122
+ manifest[newkey] = m
123
+ # Don't delete split output files - we want to keep them!
124
+ # The original parent HTML file is deleted in cleanup_section_split_instance
63
125
  end
64
126
 
65
127
  def add_section_split_instance_prep(file, key)
66
- presfile = File.join(File.dirname(@files[key][:ref]),
67
- File.basename(file[:url]))
68
- newkey = key("#{key.strip} #{file[:title]}")
128
+ # XML files are always stored flat in the _files directory (no subdirectories)
129
+ # file[:url] contains full path with directory for HTML output, but XML is basename only
130
+ xml_basename = File.basename(file[:url])
131
+ presfile = File.join(File.dirname(@files[key][:ref]), xml_basename)
132
+ newkey = entry_key("#{key.strip} #{file[:title]}")
69
133
  xml = Nokogiri::XML(File.read(presfile), &:huge)
70
134
  [presfile, newkey, xml]
71
135
  end
72
136
 
73
137
  def sectionsplit(ident)
74
138
  file = @files[ident][:ref]
139
+ # @base must always be just basename, never contain directory components
140
+ # Directory structure comes from sectionsplit_filename pattern only
141
+ base = File.basename(@files[ident][:out_path] || file)
75
142
  @sectionsplit = ::Metanorma::Collection::Sectionsplit
76
- .new(input: file, base: @files[ident][:out_path], dir: File.dirname(file),
77
- output: @files[ident][:out_path], compile_opts: @parent.compile_options,
78
- fileslookup: self, ident: ident, isodoc: @isodoc)
143
+ .new(input: file, base: base,
144
+ dir: File.dirname(file), output: @files[ident][:out_path],
145
+ compile_opts: @parent.compile_options, ident: ident,
146
+ fileslookup: self, isodoc: @isodoc,
147
+ parent_idx: @files[ident][:idx],
148
+ sectionsplit_filename: @files[ident][:sectionsplit_filename],
149
+ isodoc_presxml: @isodoc_presxml,
150
+ document_suffix: @files[ident][:document_suffix])
79
151
  coll = @sectionsplit.sectionsplit.sort_by { |f| f[:order] }
80
152
  xml = Nokogiri::XML(File.read(file, encoding: "UTF-8"), &:huge)
81
153
  [coll, @sectionsplit
@@ -0,0 +1,52 @@
1
+ module Metanorma
2
+ class Collection
3
+ class FileLookup
4
+ def anchors_lookup(anchors)
5
+ anchors.values.each_with_object({}) do |v, m|
6
+ v.each_value { |v1| m[v1] = true }
7
+ end
8
+ end
9
+
10
+ # are references to the file to be linked to a file in the collection,
11
+ # or externally? Determines whether file suffix anchors are to be used
12
+ def url?(ident)
13
+ data = get(ident) or return false
14
+ data[:url]
15
+ end
16
+
17
+ # Normalise an identifier to its @files hash key: decode entities, squeeze
18
+ # whitespace, and strip a leading "metanorma-collection " prefix. Distinct
19
+ # from Util::key, which does NOT strip that prefix.
20
+ def entry_key(ident)
21
+ @c.decode(ident).gsub(/(\p{Zs})+/, " ")
22
+ .sub(/^metanorma-collection /, "")
23
+ end
24
+
25
+ def keys
26
+ @files.keys
27
+ end
28
+
29
+ def get(ident, attr = nil)
30
+ if attr then @files[entry_key(ident)][attr]
31
+ else @files[entry_key(ident)]
32
+ end
33
+ end
34
+
35
+ def set(ident, attr, value)
36
+ @files[entry_key(ident)][attr] = value
37
+ end
38
+
39
+ def each
40
+ @files.each
41
+ end
42
+
43
+ def each_with_index
44
+ @files.each_with_index
45
+ end
46
+
47
+ def ns(xpath)
48
+ @isodoc.ns(xpath)
49
+ end
50
+ end
51
+ end
52
+ end