interscript 2.4.5 → 3.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (120) hide show
  1. checksums.yaml +4 -4
  2. data/README.md +15 -0
  3. data/docs/demo/20191118-interscript-demo-cast.gif +0 -0
  4. data/exe/codemod-imp-to-isc +5 -0
  5. data/exe/diagnose_parse_failures +47 -0
  6. data/exe/interscript +2 -1
  7. data/exe/verify_isc_deep +211 -0
  8. data/exe/verify_isc_equivalence +128 -0
  9. data/interscript.gemspec +27 -20
  10. data/lib/interscript/command.rb +21 -12
  11. data/lib/interscript/compiler/javascript.rb +34 -35
  12. data/lib/interscript/compiler/json_ir.rb +214 -0
  13. data/lib/interscript/compiler/python.rb +354 -0
  14. data/lib/interscript/compiler/ruby.rb +23 -22
  15. data/lib/interscript/compiler.rb +20 -3
  16. data/lib/interscript/detector.rb +14 -7
  17. data/lib/interscript/dsl/aliases.rb +1 -1
  18. data/lib/interscript/dsl/document.rb +6 -3
  19. data/lib/interscript/dsl/group/parallel.rb +1 -1
  20. data/lib/interscript/dsl/group.rb +18 -9
  21. data/lib/interscript/dsl/items.rb +28 -16
  22. data/lib/interscript/dsl/metadata.rb +19 -18
  23. data/lib/interscript/dsl/stage.rb +1 -1
  24. data/lib/interscript/dsl/symbol_mm.rb +4 -2
  25. data/lib/interscript/dsl/tests.rb +1 -1
  26. data/lib/interscript/dsl.rb +21 -13
  27. data/lib/interscript/interpreter.rb +32 -21
  28. data/lib/interscript/isc/codemod.rb +791 -0
  29. data/lib/interscript/isc/document_builder.rb +354 -0
  30. data/lib/interscript/isc/generator.rb +191 -0
  31. data/lib/interscript/isc/grammar/concerns/aliases.rb +29 -0
  32. data/lib/interscript/isc/grammar/concerns/dependencies.rb +23 -0
  33. data/lib/interscript/isc/grammar/concerns/items.rb +176 -0
  34. data/lib/interscript/isc/grammar/concerns/metadata.rb +142 -0
  35. data/lib/interscript/isc/grammar/concerns/primitives.rb +112 -0
  36. data/lib/interscript/isc/grammar/concerns/stages.rb +129 -0
  37. data/lib/interscript/isc/grammar/concerns/system.rb +37 -0
  38. data/lib/interscript/isc/grammar/concerns/tests.rb +31 -0
  39. data/lib/interscript/isc/grammar/concerns.rb +18 -0
  40. data/lib/interscript/isc/grammar/core.rb +23 -0
  41. data/lib/interscript/isc/grammar/isc.artifact.json +1 -0
  42. data/lib/interscript/isc/grammar/isc.parg +196 -0
  43. data/lib/interscript/isc/grammar.rb +12 -0
  44. data/lib/interscript/isc/items.rb +185 -0
  45. data/lib/interscript/isc/model/alias.rb +20 -0
  46. data/lib/interscript/isc/model/constraint.rb +20 -0
  47. data/lib/interscript/isc/model/dependency.rb +19 -0
  48. data/lib/interscript/isc/model/document.rb +31 -0
  49. data/lib/interscript/isc/model/item.rb +50 -0
  50. data/lib/interscript/isc/model/rule.rb +23 -0
  51. data/lib/interscript/isc/model/stage.rb +20 -0
  52. data/lib/interscript/isc/model/stage_item.rb +33 -0
  53. data/lib/interscript/isc/model/test.rb +21 -0
  54. data/lib/interscript/isc/model.rb +19 -0
  55. data/lib/interscript/isc/node_adapter.rb +310 -0
  56. data/lib/interscript/isc/normalizer.rb +127 -0
  57. data/lib/interscript/isc/parser.rb +78 -0
  58. data/lib/interscript/isc/serializer.rb +248 -0
  59. data/lib/interscript/isc/transform.rb +138 -0
  60. data/lib/interscript/isc/yaml_bridge.rb +236 -0
  61. data/lib/interscript/isc.rb +30 -0
  62. data/lib/interscript/ml/byt5_onnx.rb +102 -0
  63. data/lib/interscript/ml/imf.rb +182 -0
  64. data/lib/interscript/ml/model.rb +52 -0
  65. data/lib/interscript/ml/provisioning.rb +189 -0
  66. data/lib/interscript/ml/translator.rb +38 -0
  67. data/lib/interscript/ml/vocab.rb +48 -0
  68. data/lib/interscript/ml.rb +25 -0
  69. data/lib/interscript/node/alias_def.rb +5 -5
  70. data/lib/interscript/node/dependency.rb +7 -7
  71. data/lib/interscript/node/document.rb +10 -10
  72. data/lib/interscript/node/group.rb +11 -9
  73. data/lib/interscript/node/item/alias.rb +14 -10
  74. data/lib/interscript/node/item/any.rb +14 -7
  75. data/lib/interscript/node/item/capture.rb +14 -9
  76. data/lib/interscript/node/item/group.rb +22 -17
  77. data/lib/interscript/node/item/repeat.rb +3 -3
  78. data/lib/interscript/node/item/stage.rb +4 -5
  79. data/lib/interscript/node/item/string.rb +18 -13
  80. data/lib/interscript/node/item.rb +27 -17
  81. data/lib/interscript/node/metadata.rb +6 -5
  82. data/lib/interscript/node/rule/funcall.rb +4 -5
  83. data/lib/interscript/node/rule/run.rb +4 -5
  84. data/lib/interscript/node/rule/sub.rb +55 -55
  85. data/lib/interscript/node/rule.rb +6 -5
  86. data/lib/interscript/node/stage.rb +11 -14
  87. data/lib/interscript/node/tests.rb +6 -6
  88. data/lib/interscript/node.rb +14 -17
  89. data/lib/interscript/stdlib/functions/rababa_adapter.rb +56 -0
  90. data/lib/interscript/stdlib/functions/secryst_adapter.rb +37 -0
  91. data/lib/interscript/stdlib/functions.rb +49 -0
  92. data/lib/interscript/stdlib.rb +25 -95
  93. data/lib/interscript/utils/helpers.rb +8 -5
  94. data/lib/interscript/utils/regexp_converter.rb +143 -144
  95. data/lib/interscript/version.rb +1 -1
  96. data/lib/interscript/visualize/json.rb +12 -13
  97. data/lib/interscript/visualize/nodes.rb +13 -13
  98. data/lib/interscript/visualize.rb +18 -18
  99. data/lib/interscript.rb +59 -45
  100. metadata +88 -31
  101. data/.github/workflows/assets.yml +0 -91
  102. data/.github/workflows/rake.yml +0 -52
  103. data/.github/workflows/release.yml +0 -68
  104. data/.gitignore +0 -79
  105. data/.rspec +0 -3
  106. data/Gemfile +0 -38
  107. data/Rakefile +0 -139
  108. data/bin/console +0 -10
  109. data/bin/interscript +0 -5
  110. data/bin/maps_analyze_staging +0 -168
  111. data/bin/maps_debug_compilers +0 -58
  112. data/bin/maps_debug_ordering +0 -88
  113. data/bin/maps_debug_ruby_compile +0 -24
  114. data/bin/maps_debug_step_by_step +0 -44
  115. data/bin/maps_optimize_order +0 -112
  116. data/bin/maps_v1_analyze_regexps +0 -45
  117. data/bin/maps_v1_to_v2 +0 -426
  118. data/bin/set_version +0 -16
  119. data/bin/setup +0 -8
  120. data/requirements.txt +0 -1
@@ -0,0 +1,791 @@
1
+ #!/usr/bin/env ruby
2
+ # frozen_string_literal: true
3
+
4
+ require "optparse"
5
+ require "fileutils"
6
+ require "pathname"
7
+ require "strscan"
8
+
9
+ module Interscript
10
+ module Isc
11
+ # Codemod: converts a legacy `.imp` (Interscript Map Presentation) file
12
+ # into an `.isc` (Interscript/ISO Script Conversion) source file.
13
+ #
14
+ # The transformation is mechanical. It does not parse the .imp file's
15
+ # semantics — it works at the token level, applying the substitutions
16
+ # documented in <<migration-annex>> of IS 1.
17
+ #
18
+ # Usage:
19
+ # codemod-imp-to-isc <file.imp> [<file.imp>...] # convert files in place
20
+ # codemod-imp-to-isc --out-dir=DIR <file.imp>... # write to DIR
21
+ # cat foo.imp | codemod-imp-to-isc --stdin # stdin→stdout
22
+ #
23
+ class Codemod
24
+ # Compound authority segments that need hyphenation in their canonical
25
+ # form per ISO 24229 §5. The .imp filename uses the un-hyphenated
26
+ # lowercase form; the .isc system code uses the canonical hyphenated
27
+ # upper-case form.
28
+ AUTHORITY_FIXES = {
29
+ "bgnpcgn" => "BGN-PCGN",
30
+ "alalc" => "ALA-LC",
31
+ "elot" => "ELOT",
32
+ "odni" => "ODNI"
33
+ }.freeze
34
+
35
+ def initialize(out_dir: nil, stdin: false, write: true)
36
+ @out_dir = out_dir
37
+ @stdin_mode = stdin
38
+ @write = write
39
+ end
40
+
41
+ def self.run(argv)
42
+ out_dir = nil
43
+ stdin_mode = false
44
+ write = true
45
+
46
+ parser = OptionParser.new do |opts|
47
+ opts.banner = "Usage: codemod-imp-to-isc [options] <file.imp>..."
48
+ opts.on("--out-dir=DIR", "Write .isc files to DIR instead of in place") { |v| out_dir = v }
49
+ opts.on("--stdin", "Read .imp from stdin, write .isc to stdout") { stdin_mode = true }
50
+ opts.on("--dry-run", "Print converted output; do not write files") { write = false }
51
+ opts.on("-h", "--help", "Show this help") do
52
+ puts opts
53
+ exit 0
54
+ end
55
+ end
56
+ parser.parse!(argv)
57
+
58
+ cm = new(out_dir: out_dir, stdin: stdin_mode, write: write)
59
+ cm.run(argv)
60
+ end
61
+
62
+ def run(args)
63
+ if @stdin_mode
64
+ $stdout.write(convert($stdin.read, filename: "stdin"))
65
+ return
66
+ end
67
+
68
+ args.each do |path|
69
+ fail "#{path}: not a .imp file" unless path.end_with?(".imp")
70
+
71
+ source = File.read(path, encoding: "UTF-8")
72
+ converted = convert(source, filename: File.basename(path))
73
+
74
+ out_path = derive_out_path(path)
75
+ if @write
76
+ FileUtils.mkdir_p(File.dirname(out_path))
77
+ File.write(out_path, converted)
78
+ warn "#{path} -> #{out_path}"
79
+ else
80
+ $stdout.write(converted)
81
+ end
82
+ end
83
+ end
84
+
85
+ # Convert the source text of a .imp file to .isc text.
86
+ def convert(source, filename:)
87
+ @scanner = StringScanner.new(source)
88
+ @out = +""
89
+ @filename = filename
90
+
91
+ convert_body
92
+ @out
93
+ end
94
+
95
+ private
96
+
97
+ def convert_body
98
+ # 1. Emit the system wrapper, deriving the ISO 24229 code from filename.
99
+ emit_system_open
100
+
101
+ # 2. Walk the body, transforming constructs in place.
102
+ until @scanner.eos?
103
+ if @scanner.scan(/\s+/m)
104
+ @out << @scanner.matched
105
+ elsif @scanner.scan(/#[^\n]*/)
106
+ # Line comment (without consuming newline)
107
+ @out << @scanner.matched
108
+ elsif @scanner.scan(/metadata\b/)
109
+ @out << "metadata"
110
+ convert_metadata_block
111
+ elsif @scanner.scan(/tests\b/)
112
+ @out << "tests"
113
+ convert_tests_block
114
+ elsif @scanner.scan(/aliases\b/)
115
+ @out << "aliases"
116
+ convert_aliases_block
117
+ elsif @scanner.scan(/dependency\b/)
118
+ convert_dependency
119
+ elsif @scanner.scan(/stage\b/)
120
+ @out << "stage"
121
+ convert_stage_header
122
+ elsif @scanner.scan(/\b(parallel|sequence|separate|deep|compose|decompose|downcase|upcase|title_case|rababa)\b/)
123
+ @out << @scanner.matched
124
+ elsif @scanner.scan(/\bsub\b/)
125
+ @out << "sub"
126
+ convert_sub_rule
127
+ elsif @scanner.scan(/\brun\b/)
128
+ @out << "run "
129
+ convert_run_rule
130
+ elsif @scanner.scan(/\bdef_alias\b/)
131
+ convert_def_alias
132
+ elsif @scanner.scan('"')
133
+ @out << '"'
134
+ convert_string_literal(:double)
135
+ elsif @scanner.scan("'")
136
+ @out << "'"
137
+ convert_string_literal(:single)
138
+ elsif @scanner.scan("=>")
139
+ # Hash rocket — used in legacy `sub "X" => "Y"`. Convert to space.
140
+ @out << " "
141
+ elsif @scanner.scan(",")
142
+ # Trailing comma — drop in compact rule contexts, leave elsewhere.
143
+ @out << ""
144
+ elsif @scanner.scan(/(before|after|not_before|not_after|separator):/)
145
+ # Drop the colon in modifier kwarg form.
146
+ @out << "#{@scanner[1]} "
147
+ elsif @scanner.scan(/[A-Za-z_][A-Za-z0-9_]*/)
148
+ @out << @scanner.matched
149
+ else
150
+ @out << @scanner.getch
151
+ end
152
+ end
153
+
154
+ emit_system_close
155
+ end
156
+
157
+ def emit_system_open
158
+ code = derive_system_code(@filename)
159
+ @out << %(system "#{code}" {\n\n)
160
+ end
161
+
162
+ def emit_system_close
163
+ @out << "}\n"
164
+ end
165
+
166
+ def derive_system_code(filename)
167
+ stem = filename.sub(/\.imp\z/, "")
168
+ parts = stem.split("-")
169
+ authority_raw = parts.shift.to_s
170
+ authority = AUTHORITY_FIXES.fetch(authority_raw.downcase, authority_raw.upcase)
171
+ # Remaining parts: language, source_script, target_script, [year/identifying]
172
+ # Standard layout: auth-lang-src-tgt-id
173
+ language = parts.shift
174
+ source_script = parts.shift
175
+ target_script = parts.shift
176
+ identifying = parts.join("-")
177
+ # Build the source spelling: "<language>-<source_script>"
178
+ source_spelling = "#{language}-#{source_script}"
179
+ # Title-case scripts
180
+ [authority, source_spelling, target_script, identifying].compact.join(":")
181
+ end
182
+
183
+ def derive_out_path(in_path)
184
+ base = File.basename(in_path, ".imp")
185
+ new_name = "#{base}.isc"
186
+ return File.join(@out_dir, new_name) if @out_dir
187
+
188
+ File.join(File.dirname(in_path), new_name)
189
+ end
190
+
191
+ # -- Construct-specific converters
192
+
193
+ def convert_metadata_block
194
+ # Find the opening brace and consume until matching close, transforming
195
+ # `key: value` -> `key value` and `description: |` / `notes:` heredocs.
196
+ return unless @scanner.scan(/[ \t]*\{/)
197
+
198
+ @out << " {"
199
+ depth = 1
200
+
201
+ until @scanner.eos? || depth == 0
202
+ if @scanner.scan("{")
203
+ @out << "{"
204
+ depth += 1
205
+ elsif @scanner.scan("}")
206
+ depth -= 1
207
+ @out << "}"
208
+ elsif @scanner.scan(/\n[ \t]*#[^\n]*/)
209
+ # Comment line — preserve as-is
210
+ @out << @scanner.matched
211
+ elsif @scanner.scan(/\n[ \t]*\n/)
212
+ # Blank line — preserve one newline
213
+ @out << "\n"
214
+ elsif @scanner.scan(/([ \t]+)#[^\n]*/)
215
+ # Comment line (after newline was consumed by blank-line handler)
216
+ @out << "\n#{@scanner[1]}#"
217
+ elsif @scanner.scan(/#[^\n]*\n/)
218
+ # Comment at start of line
219
+ @out << "#\n"
220
+ elsif @scanner.scan(/(?:\A|\n)([ \t]+)description[ \t]*:[ \t]*\|[ \t]*\n/)
221
+ # Heredoc form: `description: |` followed by indented body.
222
+ indent = @scanner[1]
223
+ @out << "\n#{indent}description {"
224
+ convert_indented_block_until_dedent(indent)
225
+ @out << "\n#{indent}}"
226
+ elsif @scanner.scan(/(?:\A|\n)([ \t]+)description[ \t]*:[ \t]*\n[ \t]+"/)
227
+ # `description:` with multi-line QUOTED value: starts with `"` on
228
+ # next line, ends with `"` somewhere later. Capture as brace block.
229
+ indent = @scanner[1]
230
+ @out << "\n#{indent}description { "
231
+ # We've already consumed the opening `"`. Read until closing `"`.
232
+ until @scanner.eos?
233
+ if @scanner.scan(/[^"\n]+/)
234
+ @out << @scanner.matched
235
+ elsif @scanner.scan('"')
236
+ # Closing quote — don't output it (it's the YAML delimiter)
237
+ break
238
+ elsif @scanner.scan(/\n[ \t]+/)
239
+ @out << " "
240
+ elsif @scanner.scan("\n")
241
+ @out << " "
242
+ else
243
+ break
244
+ end
245
+ end
246
+ @out << " }"
247
+ elsif @scanner.scan(/(?:\A|\n)([ \t]+)description[ \t]*:[ \t]*\n(?![ \t]*[A-Za-z_]\w*[ \t]*:)(?![ \t]*\})[ \t]+/)
248
+ # `description:` with unquoted value on subsequent indented line(s).
249
+ # Negative lookahead prevents consuming next field (e.g. implementation_notes).
250
+ indent = @scanner[1]
251
+ @out << "\n#{indent}description { "
252
+ until @scanner.eos?
253
+ if @scanner.check(/\n(?:[ \t]*\n)*([ \t]{0,#{indent.length}}\S)/) ||
254
+ @scanner.check(/\n(?:[ \t]*\n)*[ \t]{0,#{indent.length}}\}/)
255
+ @out << " }"
256
+ break
257
+ elsif @scanner.scan(/[^\n]+/)
258
+ @out << @scanner.matched
259
+ elsif @scanner.scan(/\n[ \t]+/)
260
+ @out << " "
261
+ elsif @scanner.scan(/\n[ \t]*\n/)
262
+ @out << " "
263
+ elsif @scanner.scan("\n")
264
+ @out << " "
265
+ else
266
+ break
267
+ end
268
+ end
269
+ elsif @scanner.scan(/(?:\A|\n)([ \t]+)notes[ \t]*:[ \t]*\[\]/)
270
+ # `notes: []` — empty notes list
271
+ indent = @scanner[1]
272
+ @out << "\n#{indent}notes { }"
273
+ elsif @scanner.scan(/(?:\A|\n)([ \t]+)notes[ \t]*:[ \t]*""/)
274
+ # `notes: ""` — empty quoted notes value
275
+ indent = @scanner[1]
276
+ @out << "\n#{indent}notes { }"
277
+ elsif @scanner.scan(/(?:\A|\n)([ \t]+)notes[ \t]*:[ \t]*\n[ \t]+"/)
278
+ # `notes:\n "multi-line quoted value"` — quote starts on next line
279
+ indent = @scanner[1]
280
+ @out << "\n#{indent}notes {\n#{indent} note \""
281
+ convert_quoted_note_body
282
+ @out << "\"\n#{indent}}"
283
+ elsif @scanner.scan(/(?:\A|\n)([ \t]+)notes[ \t]*:[ \t]*"/)
284
+ # `notes: "X"` — single quoted-string note value
285
+ indent = @scanner[1]
286
+ @out << "\n#{indent}notes {\n#{indent} note \""
287
+ # Read until matching close quote (may span multiple lines).
288
+ convert_quoted_note_body
289
+ @out << "\"\n#{indent}}"
290
+ elsif @scanner.scan(/(?:\A|\n)([ \t]+)notes[ \t]*:[ \t]*\|[ \t]*\n/)
291
+ # Heredoc-form notes: `notes: |` followed by indented body that's
292
+ # one big multi-line note.
293
+ indent = @scanner[1]
294
+ @out << "\n#{indent}notes {"
295
+ @out << "\n#{indent} note \""
296
+ read_heredoc_into_string(indent)
297
+ @out << "\""
298
+ @out << "\n#{indent}}"
299
+ elsif @scanner.scan(/(?:\A|\n)([ \t]+)notes[ \t]*:[ \t]*/)
300
+ # Notes block: list of `- item` lines. Collect into a brace block.
301
+ indent = @scanner[1]
302
+ @out << "\n#{indent}notes {"
303
+ convert_notes_list_until_dedent(indent)
304
+ @out << "\n#{indent}}"
305
+ elsif @scanner.scan(/(?:\A|\n)([ \t]+)([A-Za-z_]\w*)[ \t]*:[ \t]*\n(?:[ \t]*#[^\n]*\n)*[ \t]*\n*([ \t]+)-[ \t]*/)
306
+ # Multi-line list value: `field:\n [optional comments]\n [optional blank]\n - item`
307
+ indent = @scanner[1]
308
+ field = @scanner[2]
309
+ item_indent = @scanner[3]
310
+ @out << "\n#{indent}#{field} {"
311
+ @out << "\n#{item_indent}- "
312
+ text = @scanner.scan(/[^\n]+/).to_s
313
+ @out << escape_braces(text)
314
+ convert_indented_block_until_dedent(indent)
315
+ @out << "\n#{indent}}"
316
+ elsif @scanner.scan(/(?:\A|\n)([ \t]+)([A-Za-z_]\w*)[ \t]*:[ \t]*\n([ \t]+)(?![ \t]*(?:-|"|\[|\]|\|))(?![ \t]*$)(?![ \t]*[A-Za-z_]\w*[ \t]*:)/)
317
+ # Multi-line unquoted text value: `field:\n text` (not list, quote,
318
+ # heredoc, or another field declaration at the same indent)
319
+ indent = @scanner[1]
320
+ field = @scanner[2]
321
+ @out << "\n#{indent}#{field} {"
322
+ convert_indented_block_until_dedent(indent)
323
+ @out << "\n#{indent}}"
324
+ elsif @scanner.scan(/(?:\A|\n)([ \t]+)([A-Za-z_]\w*)[ \t]*:[ \t]*\|[ \t]*\n/)
325
+ # Generic field with heredoc: `field: |\n body`
326
+ indent = @scanner[1]
327
+ field = @scanner[2]
328
+ @out << "\n#{indent}#{field} { "
329
+ convert_indented_block_until_dedent(indent)
330
+ @out << " }"
331
+ elsif @scanner.scan(/(?:\A|\n)([ \t]+)([A-Za-z_]\w*)[ \t]*:[ \t]*/)
332
+ # key: value -> key value, only when the key is at the start of a
333
+ # (indented) line. Use [ \t] instead of \s to avoid eating newlines.
334
+ @out << "\n#{@scanner[1]}#{@scanner[2]} "
335
+ elsif @scanner.scan('"')
336
+ @out << '"'
337
+ elsif @scanner.scan("'")
338
+ @out << "'"
339
+ else
340
+ @out << @scanner.getch
341
+ end
342
+ end
343
+ end
344
+
345
+ # Consume an indented heredoc body until a non-blank line dedents at or
346
+ # below `indent`. Blank lines within the body are preserved.
347
+ def convert_indented_block_until_dedent(indent)
348
+ # Look ahead through optional blank lines: if the next non-blank line
349
+ # is dedented to indent depth <= indent.length, the heredoc ends.
350
+ dedent_check = /\n(?:[ \t]*\n)*([ \t]{0,#{indent.length}}\S)/
351
+
352
+ until @scanner.eos?
353
+ if @scanner.check(dedent_check)
354
+ return
355
+ elsif @scanner.scan(/\n[ \t]*\n/)
356
+ @out << @scanner.matched
357
+ elsif @scanner.scan(/\n([ \t]+)/)
358
+ @out << "\n#{@scanner[1]}"
359
+ elsif @scanner.scan("\n")
360
+ @out << "\n"
361
+ elsif @scanner.scan(/[^\n]+/)
362
+ @out << escape_braces(@scanner.matched)
363
+ else
364
+ @out << @scanner.getch
365
+ end
366
+ end
367
+ end
368
+
369
+ def escape_braces(text)
370
+ text.gsub("\\", "\\\\\\\\").gsub(/[{}]/) { |c| "\\#{c}" }
371
+ end
372
+
373
+ # Notes list: each item begins with `- `. Convert each to `note "..."`.
374
+ # A `- |` item is a multi-line YAML heredoc; consume subsequent indented lines.
375
+ def convert_notes_list_until_dedent(indent)
376
+ # A dedent is: a non-blank line whose first non-whitespace char is at
377
+ # indent depth <= indent.length AND isn't a `-` list marker (which
378
+ # would be another note at the same indent).
379
+ dedent_check = /\n(?:[ \t]*\n)*([ \t]{0,#{indent.length}}[^-\s])/
380
+
381
+ loop do
382
+ break if @scanner.eos?
383
+
384
+ if @scanner.check(dedent_check)
385
+ return
386
+ elsif @scanner.check(/\n(?:[ \t]*\n)*[ \t]{0,#{indent.length}}\}/)
387
+ # Hit the enclosing metadata `}` (possibly after blank lines).
388
+ return
389
+ elsif @scanner.scan(/\n[ \t]*\n/)
390
+ # Blank line(s) — preserve one newline. Do NOT consume the
391
+ # indent of the next item.
392
+ @out << "\n"
393
+ elsif @scanner.scan(/\n([ \t]+)-[ \t]*\|(?:[ \t]*#[^\n]*)?\n/)
394
+ # `|` heredoc form (with optional inline comment after |)
395
+ note_indent = @scanner[1]
396
+ @out << "\n#{note_indent}note \""
397
+ read_heredoc_into_string(note_indent)
398
+ @out << "\""
399
+ elsif @scanner.scan(/\n([ \t]+)-[ \t]+/)
400
+ # Single-line item start (possibly with continuation lines).
401
+ note_indent = @scanner[1]
402
+ emit_note_with_continuation(note_indent)
403
+ elsif @scanner.scan(/([ \t]+)-[ \t]*\|(?:[ \t]*#[^\n]*)?\n/)
404
+ # First item right after `notes:` consumed; scanner at `<indent>- |\n`.
405
+ emit_heredoc_note(@scanner[1])
406
+ elsif @scanner.scan(/([ \t]+)-[ \t]+/)
407
+ # First item right after `notes:` consumed; scanner at `<indent>- item`.
408
+ emit_note_with_continuation(@scanner[1])
409
+ elsif @scanner.scan("\n")
410
+ @out << "\n"
411
+ else
412
+ @out << @scanner.getch
413
+ end
414
+ end
415
+ end
416
+
417
+ def emit_heredoc_note(indent)
418
+ @out << "\n#{indent}note \""
419
+ read_heredoc_into_string(indent)
420
+ @out << "\""
421
+ end
422
+
423
+ def emit_note_with_continuation(note_indent)
424
+ @out << "\n#{note_indent}note \""
425
+ text = @scanner.scan(/[^\n]+/).to_s
426
+ # Strip YAML inline comments: "text # comment" → "text"
427
+ text = text.sub(/\s+#.*$/, "")
428
+ # Track if item was YAML-quoted (for trailing quote cleanup)
429
+ was_dquote = text.start_with?('"') && !text.end_with?('"')
430
+ was_squote = text.start_with?("'") && !text.end_with?("'")
431
+ # Strip outer quotes if the YAML list item was quoted: - "text"
432
+ text = text[1..-2] if text.start_with?('"') && text.end_with?('"')
433
+ text = text[1..-2] if text.start_with?("'") && text.end_with?("'")
434
+ # Also strip leading quote when text spans multiple lines (closing on later line)
435
+ text = text[1..] if text.start_with?('"') && !text.end_with?('"')
436
+ text = text[1..] if text.start_with?("'") && !text.end_with?("'")
437
+ # Unescape YAML escape sequences, then re-escape for ISC
438
+ text = text.gsub('\\"', '"').gsub("\\\\", "\\")
439
+ @out << text.gsub("\\", "\\\\\\\\").gsub('"', '\\"').gsub("\\u", "\\\\\\\\u")
440
+ # Consume continuation lines: any subsequent line indented deeper
441
+ # than the `- ` marker is part of the same note. Blank lines between
442
+ # continuations are preserved as \n.
443
+ loop do
444
+ if @scanner.check(/\n[ \t]{#{note_indent.length + 1},}\S/)
445
+ # Indented continuation
446
+ @scanner.scan(/\n([ \t]+)/)
447
+ @out << "\\n" + @scanner[1].strip + " "
448
+ cont = @scanner.scan(/[^\n]+/).to_s
449
+ cont = cont.gsub("\\", "\\\\\\\\").gsub('"', '\\"').gsub("\\u", "\\\\\\\\u")
450
+ @out << cont
451
+ elsif @scanner.check(/\n[ \t]*\n[ \t]{#{note_indent.length + 1},}\S/)
452
+ # Blank line then indented continuation
453
+ @scanner.scan(/\n[ \t]*\n([ \t]+)/)
454
+ @out << "\\n" + @scanner[1].strip + " "
455
+ cont = @scanner.scan(/[^\n]+/).to_s
456
+ cont = cont.gsub("\\", "\\\\\\\\").gsub('"', '\\"').gsub("\\u", "\\\\\\\\u")
457
+ @out << cont
458
+ else
459
+ break
460
+ end
461
+ end
462
+ # If item was multi-line YAML-quoted, strip the trailing closing
463
+ # quote that leaked from the last continuation line.
464
+ if was_dquote && @out.end_with?('\\"')
465
+ @out[-2..] = ""
466
+ elsif was_squote && @out.end_with?("'")
467
+ @out[-1..] = ""
468
+ end
469
+ @out << "\""
470
+ end
471
+
472
+ def convert_quoted_note_body
473
+ # Read a quoted string body (already past opening quote). Continues
474
+ # across newlines until matching unescaped `"`.
475
+ until @scanner.eos?
476
+ if @scanner.scan(/\\./)
477
+ @out << @scanner.matched
478
+ elsif @scanner.scan('"')
479
+ return
480
+ else
481
+ c = @scanner.getch
482
+ @out << ((c == "\n") ? "\\n" : c)
483
+ end
484
+ end
485
+ end
486
+
487
+ def read_heredoc_into_string(indent)
488
+ # Read lines that are indented deeper than `indent` (or blank). Concatenate.
489
+ # Preserve raw indentation — the DocumentBuilder's normalize_heredoc
490
+ # handles YAML-style dedent to match the DSL's output.
491
+ until @scanner.eos?
492
+ if @scanner.check(/\n(?:[ \t]*\n)*([ \t]{0,#{indent.length}}\S)/)
493
+ return
494
+ elsif @scanner.scan(/\n[ \t]*\n/)
495
+ # Blank line inside heredoc — preserve as \n\n
496
+ @out << "\\n\\n"
497
+ elsif @scanner.scan(/\n([ \t]+[^\n]*)/)
498
+ # Indented line — preserve raw content (indent + text)
499
+ line = @scanner[1].to_s.gsub('"', '\\"').gsub("\\u", "\\\\\\\\u")
500
+ @out << "\\n" + line
501
+ elsif @scanner.scan("\n")
502
+ @out << "\\n"
503
+ elsif @scanner.scan(/([^\n]+)/)
504
+ line = @scanner[1].gsub('"', '\\"').gsub("\\u", "\\\\\\\\u")
505
+ @out << line
506
+ else
507
+ return
508
+ end
509
+ end
510
+ end
511
+
512
+ def convert_tests_block
513
+ return unless @scanner.scan(/[ \t]*\{/)
514
+ @out << " {"
515
+ depth = 1
516
+ until @scanner.eos? || depth == 0
517
+ if @scanner.scan("{")
518
+ @out << "{"
519
+ depth += 1
520
+ elsif @scanner.scan("}")
521
+ depth -= 1
522
+ @out << "}"
523
+ elsif @scanner.scan(/#[^\n]*/)
524
+ # Preserve comment lines verbatim
525
+ @out << @scanner.matched
526
+ elsif @scanner.scan(/\btest\b/)
527
+ # `test "X", "Y"` -> `"X" -> "Y"`
528
+ @out << ""
529
+ elsif @scanner.scan(",")
530
+ # Comma between test args -> ` -> `
531
+ @out << " -> "
532
+ elsif @scanner.scan('"')
533
+ @out << '"'
534
+ convert_string_literal(:double)
535
+ elsif @scanner.scan("'")
536
+ @out << "'"
537
+ convert_string_literal(:single)
538
+ else
539
+ @out << @scanner.getch
540
+ end
541
+ end
542
+ end
543
+
544
+ def convert_aliases_block
545
+ return unless @scanner.scan(/[ \t]*\{/)
546
+ @out << " {"
547
+ depth = 1
548
+ until @scanner.eos? || depth == 0
549
+ if @scanner.scan("{")
550
+ @out << "{"
551
+ depth += 1
552
+ elsif @scanner.scan("}")
553
+ depth -= 1
554
+ @out << "}"
555
+ elsif @scanner.scan(/(?:\A|\n)([ \t]+)def_alias\s+([A-Za-z_]\w*)\s*,\s*/)
556
+ # Legacy `def_alias name, X` -> `name = X`. Capture indent + name.
557
+ indent = @scanner[1]
558
+ name = @scanner[2]
559
+ @out << "\n#{indent}#{name} = "
560
+ elsif @scanner.scan(/def_alias\s+([A-Za-z_]\w*)\s*,\s*/)
561
+ # `def_alias name, X` at start of aliases block (no leading newline)
562
+ @out << "#{@scanner[1]} = "
563
+ elsif @scanner.scan('"')
564
+ @out << '"'
565
+ convert_string_literal(:double)
566
+ elsif @scanner.scan("'")
567
+ @out << "'"
568
+ convert_string_literal(:single)
569
+ elsif @scanner.scan(/#[^\n]*/)
570
+ @out << @scanner.matched
571
+ else
572
+ @out << @scanner.getch
573
+ end
574
+ end
575
+ end
576
+
577
+ def convert_dependency
578
+ # Forms accepted:
579
+ # dependency "X" -> dependency "X"
580
+ # dependency "X", as: Y -> dependency "X" as Y
581
+ # dependency "X", import: true -> dependency "X" (import dropped; isc imports via `run`)
582
+ # dependency "X", as: Y, import: true -> dependency "X" as Y
583
+ @out << "dependency"
584
+ # Consume up to end of line/statement, handling modifiers.
585
+ until @scanner.eos?
586
+ if @scanner.scan(/,?\s*as\s*:\s*/)
587
+ @out << " as "
588
+ # Read the alias identifier
589
+ @scanner.scan(/[A-Za-z_]\w*/) && @out << @scanner.matched
590
+ # Continue past this; may have more modifiers
591
+ elsif @scanner.scan(/,?\s*import\s*:\s*true/)
592
+ # Drop `import: true` — isc handles imports via `run map.X.stage.Y`.
593
+ # No output.
594
+ elsif @scanner.scan(/[\n}]/)
595
+ @scanner.unscan
596
+ return
597
+ elsif @scanner.scan('"')
598
+ @out << '"'
599
+ convert_string_literal(:double)
600
+ elsif @scanner.scan(/[^\n",}]+/)
601
+ @out << @scanner.matched
602
+ else
603
+ # No progress — bail to avoid infinite loop.
604
+ @scanner.getch
605
+ end
606
+ end
607
+ end
608
+
609
+ def convert_stage_header
610
+ # Legacy `stage {` becomes `stage main {` if no name is given.
611
+ # `stage(translit) {` becomes `stage translit {`.
612
+ if @scanner.scan(/\s*\(([A-Za-z_]\w*)\)\s*\{/)
613
+ @out << " #{@scanner[1]} {"
614
+ elsif @scanner.scan(/[ \t]*\{/)
615
+ @out << " main {"
616
+ elsif @scanner.scan(/\s+([A-Za-z_]\w*)\s*\{/)
617
+ @out << " #{@scanner[1]} {"
618
+ end
619
+ end
620
+
621
+ def convert_sub_rule
622
+ # Read the rule's from, to, and optional constraints from the source.
623
+ # The .imp form is one of:
624
+ # sub "X", "Y", before: Z (positional + kwargs)
625
+ # sub "X" => "Y", before: Z (hash rocket)
626
+ # sub "X", "Y" (no constraints)
627
+ # sub "X" + any(Y), "Z", before: W (concat in from)
628
+ #
629
+ # Output: if from/to are simple (single quoted string or atom each),
630
+ # emit compact form `sub "X" "Y"`. Otherwise emit block form:
631
+ # sub {
632
+ # from <expr>
633
+ # to <expr>
634
+ # before <expr>
635
+ # ...
636
+ # }
637
+
638
+ # Tokenize the rule body up to the next `\n` (rules are single-line)
639
+ # or unindented `}`. Capture: from_expr, comma, to_expr, constraints.
640
+ from_expr, to_expr, constraints_str = tokenize_sub_rule
641
+
642
+ # Decide compact vs block form.
643
+ compact_safe = single_atom?(from_expr) && single_atom?(to_expr) && constraints_str.empty?
644
+
645
+ if compact_safe
646
+ @out << " #{from_expr} #{to_expr}\n"
647
+ else
648
+ @out << " {\n"
649
+ @out << " from #{from_expr}\n" unless from_expr.empty?
650
+ @out << " to #{to_expr}\n" unless to_expr.empty?
651
+ unless constraints_str.empty?
652
+ constraints_str.strip.split(/(?=\b(?:before|after|not_before|not_after)\b)/).each do |c|
653
+ @out << " #{c.strip}\n" unless c.strip.empty?
654
+ end
655
+ end
656
+ @out << " }\n"
657
+ end
658
+ end
659
+
660
+ # Tokenize a sub rule body. Returns [from, to, constraints_string].
661
+ # Advances the scanner past the rule (consumes up to and including the
662
+ # trailing newline).
663
+ def tokenize_sub_rule
664
+ # Read until end of line. Rules are single-line in .imp.
665
+ line = @scanner.scan_until(/\n/).to_s
666
+ # Drop the trailing newline
667
+ line = line.chomp
668
+
669
+ # Strip comments (# ... to end of line) but only when # is at start of
670
+ # token (not inside a string). Walk char by char.
671
+ line = strip_comments(line)
672
+
673
+ # Split into tokens: handle hash rockets, commas, parens, strings.
674
+ tokens = []
675
+ current = +""
676
+ in_string = nil
677
+ paren_depth = 0
678
+
679
+ line.each_char.with_index do |c, i|
680
+ if in_string
681
+ current << c
682
+ if c == in_string && current[-2] != "\\"
683
+ in_string = nil
684
+ end
685
+ elsif c == '"' || c == "'"
686
+ in_string = c
687
+ current << c
688
+ elsif c == "("
689
+ paren_depth += 1
690
+ current << c
691
+ elsif c == ")"
692
+ paren_depth -= 1
693
+ current << c
694
+ elsif paren_depth.zero? && (c == "," || (c == "=" && line[i + 1] == ">"))
695
+ tokens << current.strip
696
+ current = +""
697
+ else
698
+ current << c
699
+ end
700
+ end
701
+ tokens << current.strip unless current.strip.empty?
702
+
703
+ tokens = tokens.reject { |t| t == "=>" }
704
+
705
+ from_expr = normalize_expr(tokens.shift.to_s)
706
+ to_expr = normalize_expr(tokens.shift.to_s)
707
+ constraints_str = tokens.join(" ")
708
+
709
+ constraints_str = constraints_str.gsub(/(before|after|not_before|not_after)\s*:/, '\1')
710
+
711
+ [from_expr, to_expr, constraints_str]
712
+ end
713
+
714
+ # Remove `# ...` comments from a line, respecting quoted strings.
715
+ def strip_comments(line)
716
+ result = +""
717
+ in_string = nil
718
+ line.each_char do |c|
719
+ if in_string
720
+ result << c
721
+ if c == in_string && result[-2] != "\\"
722
+ in_string = nil
723
+ end
724
+ elsif c == '"' || c == "'"
725
+ in_string = c
726
+ result << c
727
+ elsif c == "#"
728
+ break
729
+ else
730
+ result << c
731
+ end
732
+ end
733
+ result
734
+ end
735
+
736
+ # A "single atom" expression is one quoted string, `none`, `boundary`,
737
+ # `line_start`, `line_end`, `word_boundary`, or a bare alias identifier.
738
+ # Anything with `+`, `any(`, `capture(`, `maybe(`, or concatenation is
739
+ # NOT a single atom.
740
+ def single_atom?(expr)
741
+ return false if expr.nil? || expr.empty?
742
+ return false if expr.include?("+")
743
+ return false if /\b(any|capture|maybe)\s*\(/.match?(expr)
744
+ s = expr.strip
745
+ return true if s =~ /\A"[^"]*"\z/ || s =~ /\A'[^']*'\z/
746
+ return true if ["none", "boundary", "line_start", "line_end", "word_boundary"].include?(s)
747
+ return true if /\A[a-zA-Z_][a-zA-Z0-9_]*\z/.match?(s)
748
+ false
749
+ end
750
+
751
+ # Normalize a captured expression: drop redundant whitespace around
752
+ # `+` operators. `sub "X" , "Y"` -> tokens ["\"X\"", "\"Y\""].
753
+ def normalize_expr(expr)
754
+ expr = expr.strip
755
+ # Collapse runs of whitespace
756
+ expr = expr.gsub(/\s+/, " ")
757
+ # Remove space around +
758
+ expr.gsub(/\s*\+\s*/, " + ")
759
+ end
760
+
761
+ def convert_run_rule
762
+ # `run map.X.stage.Y` -> preserved
763
+ # `run stage.Y` -> preserved (without map. prefix)
764
+ # `run map.X.stage(Y)` -> `run map.X.stage.Y`
765
+ if @scanner.scan(/map\.([A-Za-z_]\w*)\.stage\.([A-Za-z_]\w*)/)
766
+ @out << "map.#{@scanner[1]}.stage.#{@scanner[2]}"
767
+ elsif @scanner.scan(/stage\.([A-Za-z_]\w*)/)
768
+ @out << "stage.#{@scanner[1]}"
769
+ end
770
+ end
771
+
772
+ def convert_def_alias
773
+ # Handled in convert_aliases_block.
774
+ end
775
+
776
+ def convert_string_literal(quote_kind)
777
+ quote_char = (quote_kind == :double) ? '"' : "'"
778
+ until @scanner.eos?
779
+ if @scanner.scan(/\\./)
780
+ @out << @scanner.matched
781
+ elsif @scanner.scan(Regexp.new(Regexp.escape(quote_char)))
782
+ @out << quote_char
783
+ return
784
+ else
785
+ @out << @scanner.getch
786
+ end
787
+ end
788
+ end
789
+ end
790
+ end
791
+ end