interscript 2.4.5 → 3.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +15 -0
- data/docs/demo/20191118-interscript-demo-cast.gif +0 -0
- data/exe/codemod-imp-to-isc +5 -0
- data/exe/diagnose_parse_failures +47 -0
- data/exe/interscript +2 -1
- data/exe/verify_isc_deep +211 -0
- data/exe/verify_isc_equivalence +128 -0
- data/interscript.gemspec +27 -20
- data/lib/interscript/command.rb +21 -12
- data/lib/interscript/compiler/javascript.rb +34 -35
- data/lib/interscript/compiler/json_ir.rb +214 -0
- data/lib/interscript/compiler/python.rb +354 -0
- data/lib/interscript/compiler/ruby.rb +23 -22
- data/lib/interscript/compiler.rb +20 -3
- data/lib/interscript/detector.rb +14 -7
- data/lib/interscript/dsl/aliases.rb +1 -1
- data/lib/interscript/dsl/document.rb +6 -3
- data/lib/interscript/dsl/group/parallel.rb +1 -1
- data/lib/interscript/dsl/group.rb +18 -9
- data/lib/interscript/dsl/items.rb +28 -16
- data/lib/interscript/dsl/metadata.rb +19 -18
- data/lib/interscript/dsl/stage.rb +1 -1
- data/lib/interscript/dsl/symbol_mm.rb +4 -2
- data/lib/interscript/dsl/tests.rb +1 -1
- data/lib/interscript/dsl.rb +21 -13
- data/lib/interscript/interpreter.rb +32 -21
- data/lib/interscript/isc/codemod.rb +791 -0
- data/lib/interscript/isc/document_builder.rb +354 -0
- data/lib/interscript/isc/generator.rb +191 -0
- data/lib/interscript/isc/grammar/concerns/aliases.rb +29 -0
- data/lib/interscript/isc/grammar/concerns/dependencies.rb +23 -0
- data/lib/interscript/isc/grammar/concerns/items.rb +176 -0
- data/lib/interscript/isc/grammar/concerns/metadata.rb +142 -0
- data/lib/interscript/isc/grammar/concerns/primitives.rb +112 -0
- data/lib/interscript/isc/grammar/concerns/stages.rb +129 -0
- data/lib/interscript/isc/grammar/concerns/system.rb +37 -0
- data/lib/interscript/isc/grammar/concerns/tests.rb +31 -0
- data/lib/interscript/isc/grammar/concerns.rb +18 -0
- data/lib/interscript/isc/grammar/core.rb +23 -0
- data/lib/interscript/isc/grammar/isc.artifact.json +1 -0
- data/lib/interscript/isc/grammar/isc.parg +196 -0
- data/lib/interscript/isc/grammar.rb +12 -0
- data/lib/interscript/isc/items.rb +185 -0
- data/lib/interscript/isc/model/alias.rb +20 -0
- data/lib/interscript/isc/model/constraint.rb +20 -0
- data/lib/interscript/isc/model/dependency.rb +19 -0
- data/lib/interscript/isc/model/document.rb +31 -0
- data/lib/interscript/isc/model/item.rb +50 -0
- data/lib/interscript/isc/model/rule.rb +23 -0
- data/lib/interscript/isc/model/stage.rb +20 -0
- data/lib/interscript/isc/model/stage_item.rb +33 -0
- data/lib/interscript/isc/model/test.rb +21 -0
- data/lib/interscript/isc/model.rb +19 -0
- data/lib/interscript/isc/node_adapter.rb +310 -0
- data/lib/interscript/isc/normalizer.rb +127 -0
- data/lib/interscript/isc/parser.rb +78 -0
- data/lib/interscript/isc/serializer.rb +248 -0
- data/lib/interscript/isc/transform.rb +138 -0
- data/lib/interscript/isc/yaml_bridge.rb +236 -0
- data/lib/interscript/isc.rb +30 -0
- data/lib/interscript/ml/byt5_onnx.rb +102 -0
- data/lib/interscript/ml/imf.rb +182 -0
- data/lib/interscript/ml/model.rb +52 -0
- data/lib/interscript/ml/provisioning.rb +189 -0
- data/lib/interscript/ml/translator.rb +38 -0
- data/lib/interscript/ml/vocab.rb +48 -0
- data/lib/interscript/ml.rb +25 -0
- data/lib/interscript/node/alias_def.rb +5 -5
- data/lib/interscript/node/dependency.rb +7 -7
- data/lib/interscript/node/document.rb +10 -10
- data/lib/interscript/node/group.rb +11 -9
- data/lib/interscript/node/item/alias.rb +14 -10
- data/lib/interscript/node/item/any.rb +14 -7
- data/lib/interscript/node/item/capture.rb +14 -9
- data/lib/interscript/node/item/group.rb +22 -17
- data/lib/interscript/node/item/repeat.rb +3 -3
- data/lib/interscript/node/item/stage.rb +4 -5
- data/lib/interscript/node/item/string.rb +18 -13
- data/lib/interscript/node/item.rb +27 -17
- data/lib/interscript/node/metadata.rb +6 -5
- data/lib/interscript/node/rule/funcall.rb +4 -5
- data/lib/interscript/node/rule/run.rb +4 -5
- data/lib/interscript/node/rule/sub.rb +55 -55
- data/lib/interscript/node/rule.rb +6 -5
- data/lib/interscript/node/stage.rb +11 -14
- data/lib/interscript/node/tests.rb +6 -6
- data/lib/interscript/node.rb +14 -17
- data/lib/interscript/stdlib/functions/rababa_adapter.rb +56 -0
- data/lib/interscript/stdlib/functions/secryst_adapter.rb +37 -0
- data/lib/interscript/stdlib/functions.rb +49 -0
- data/lib/interscript/stdlib.rb +25 -95
- data/lib/interscript/utils/helpers.rb +8 -5
- data/lib/interscript/utils/regexp_converter.rb +143 -144
- data/lib/interscript/version.rb +1 -1
- data/lib/interscript/visualize/json.rb +12 -13
- data/lib/interscript/visualize/nodes.rb +13 -13
- data/lib/interscript/visualize.rb +18 -18
- data/lib/interscript.rb +59 -45
- metadata +88 -31
- data/.github/workflows/assets.yml +0 -91
- data/.github/workflows/rake.yml +0 -52
- data/.github/workflows/release.yml +0 -68
- data/.gitignore +0 -79
- data/.rspec +0 -3
- data/Gemfile +0 -38
- data/Rakefile +0 -139
- data/bin/console +0 -10
- data/bin/interscript +0 -5
- data/bin/maps_analyze_staging +0 -168
- data/bin/maps_debug_compilers +0 -58
- data/bin/maps_debug_ordering +0 -88
- data/bin/maps_debug_ruby_compile +0 -24
- data/bin/maps_debug_step_by_step +0 -44
- data/bin/maps_optimize_order +0 -112
- data/bin/maps_v1_analyze_regexps +0 -45
- data/bin/maps_v1_to_v2 +0 -426
- data/bin/set_version +0 -16
- data/bin/setup +0 -8
- data/requirements.txt +0 -1
|
@@ -0,0 +1,176 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "parsanol"
|
|
4
|
+
|
|
5
|
+
module Interscript
|
|
6
|
+
module Isc
|
|
7
|
+
module Grammar
|
|
8
|
+
module Concerns
|
|
9
|
+
# Item expressions: the building blocks of rule matches and targets.
|
|
10
|
+
module Items
|
|
11
|
+
include Parsanol
|
|
12
|
+
|
|
13
|
+
rule(:item_atom) do
|
|
14
|
+
quoted_string |
|
|
15
|
+
str("none").as(:none) |
|
|
16
|
+
zero_width_primitive |
|
|
17
|
+
any_character |
|
|
18
|
+
any_constructor |
|
|
19
|
+
capture_constructor |
|
|
20
|
+
maybe_constructor |
|
|
21
|
+
some_constructor |
|
|
22
|
+
function_call |
|
|
23
|
+
capture_reference |
|
|
24
|
+
alias_reference
|
|
25
|
+
end
|
|
26
|
+
|
|
27
|
+
# some(...) — one or more matches (greedy).
|
|
28
|
+
rule(:some_constructor) do
|
|
29
|
+
str("some") >> str("(") >> whitespace? >>
|
|
30
|
+
item.as(:some_inner) >>
|
|
31
|
+
whitespace? >> str(")")
|
|
32
|
+
end
|
|
33
|
+
|
|
34
|
+
# Function call: upcase, downcase, title_case, reverse, etc.
|
|
35
|
+
# These appear as the `to` value in sub rules: `sub "X" upcase`.
|
|
36
|
+
rule(:function_call) do
|
|
37
|
+
(str("upcase") | str("downcase") | str("title_case") |
|
|
38
|
+
str("reverse") | str("strip") | str("swapcase")).as(:function)
|
|
39
|
+
end
|
|
40
|
+
|
|
41
|
+
rule(:zero_width_primitive) do
|
|
42
|
+
(
|
|
43
|
+
str("boundary") |
|
|
44
|
+
str("line_start") |
|
|
45
|
+
str("line_end") |
|
|
46
|
+
str("word_boundary") |
|
|
47
|
+
str("space") |
|
|
48
|
+
str("non_boundary")
|
|
49
|
+
).as(:primitive)
|
|
50
|
+
end
|
|
51
|
+
|
|
52
|
+
# any_character — matches any single character.
|
|
53
|
+
rule(:any_character) do
|
|
54
|
+
str("any_character").as(:any_char)
|
|
55
|
+
end
|
|
56
|
+
|
|
57
|
+
rule(:any_constructor) do
|
|
58
|
+
str("any") >> str("(") >> whitespace? >>
|
|
59
|
+
(range_arg | set_arg | alias_arg | item.as(:any_item)).as(:any) >>
|
|
60
|
+
whitespace? >> str(")")
|
|
61
|
+
end
|
|
62
|
+
|
|
63
|
+
# `any(identifier)` — accept a bare alias reference inside any().
|
|
64
|
+
# Exclude zero-width primitives (space, boundary, etc.) which are
|
|
65
|
+
# handled by `item` via `zero_width_primitive` in `item_atom`.
|
|
66
|
+
rule(:alias_arg) do
|
|
67
|
+
(zero_width_primitive.absent? >> keyword.absent? >> identifier).as(:alias_ref)
|
|
68
|
+
end
|
|
69
|
+
|
|
70
|
+
# capture(...) — wraps a sub-expression with a capture group.
|
|
71
|
+
# The captured value can be referenced in the target via `ref(N)`.
|
|
72
|
+
rule(:capture_constructor) do
|
|
73
|
+
str("capture") >> str("(") >> whitespace? >>
|
|
74
|
+
item.as(:capture_inner) >>
|
|
75
|
+
whitespace? >> str(")")
|
|
76
|
+
end
|
|
77
|
+
|
|
78
|
+
# maybe(...) — optional match (zero or one occurrence).
|
|
79
|
+
rule(:maybe_constructor) do
|
|
80
|
+
str("maybe") >> str("(") >> whitespace? >>
|
|
81
|
+
item.as(:maybe_inner) >>
|
|
82
|
+
whitespace? >> str(")")
|
|
83
|
+
end
|
|
84
|
+
|
|
85
|
+
rule(:range_arg) do
|
|
86
|
+
quoted_string.as(:lo) >>
|
|
87
|
+
whitespace? >> str("..") >> whitespace? >>
|
|
88
|
+
quoted_string.as(:hi)
|
|
89
|
+
end
|
|
90
|
+
|
|
91
|
+
rule(:set_arg) do
|
|
92
|
+
quoted_string.as(:single) |
|
|
93
|
+
(str("[") >> whitespace? >>
|
|
94
|
+
(list_item >> ((comma | whitespace) >> list_item).repeat).as(:list) >>
|
|
95
|
+
whitespace? >> str("]"))
|
|
96
|
+
end
|
|
97
|
+
|
|
98
|
+
# A list item is an item expression (which includes quoted strings).
|
|
99
|
+
rule(:list_item) do
|
|
100
|
+
item
|
|
101
|
+
end
|
|
102
|
+
|
|
103
|
+
rule(:alias_reference) do
|
|
104
|
+
(keyword.absent? >> identifier >>
|
|
105
|
+
(str(".") >> identifier.as(:qualified_name)).maybe >>
|
|
106
|
+
str("{").absent?).as(:alias)
|
|
107
|
+
end
|
|
108
|
+
|
|
109
|
+
# ref(N) — reference to Nth capture group. Only valid in `to` position.
|
|
110
|
+
rule(:capture_reference) do
|
|
111
|
+
(str("ref") >> str("(") >> whitespace? >>
|
|
112
|
+
match(/[0-9]/).as(:digit) >> whitespace? >>
|
|
113
|
+
str(")")).as(:ref)
|
|
114
|
+
end
|
|
115
|
+
|
|
116
|
+
rule(:keyword) do
|
|
117
|
+
str("parallel") | str("sequence") | str("stage") |
|
|
118
|
+
str("compose") | str("separate") | str("system") |
|
|
119
|
+
str("metadata") | str("aliases") | str("tests") |
|
|
120
|
+
str("notes") | str("description") | str("name") |
|
|
121
|
+
str("authority") | str("dependency") | str("run") |
|
|
122
|
+
str("sub") | str("before") | str("after") |
|
|
123
|
+
str("not_before") | str("not_after") | str("any") |
|
|
124
|
+
str("none") | str("boundary") | str("line_start") |
|
|
125
|
+
str("line_end") | str("word_boundary") |
|
|
126
|
+
str("downcase") | str("upcase") | str("title_case") |
|
|
127
|
+
str("capture") | str("maybe") | str("some") | str("ref") |
|
|
128
|
+
str("any_character")
|
|
129
|
+
end
|
|
130
|
+
|
|
131
|
+
# Concatenation: one or more atoms. The continuation pattern
|
|
132
|
+
# requires that whitespace or `+` be IMMEDIATELY followed by
|
|
133
|
+
# something that's clearly an item_atom start (a quote, `(`,
|
|
134
|
+
# letter, etc.) AND not a block-rule keyword like `to`, `before`.
|
|
135
|
+
rule(:item) do
|
|
136
|
+
(item_atom >>
|
|
137
|
+
((concat_sep >> item_continuation.present?) >> item_atom).repeat
|
|
138
|
+
).as(:concatenation)
|
|
139
|
+
end
|
|
140
|
+
|
|
141
|
+
rule(:concat_sep) do
|
|
142
|
+
(whitespace? >> str("+") >> whitespace?) | whitespace
|
|
143
|
+
end
|
|
144
|
+
|
|
145
|
+
# Positive lookahead: the next thing is a valid item_atom continuation.
|
|
146
|
+
# Excludes keywords that end the item (to, before, after, not_before,
|
|
147
|
+
# not_after, the closing brace, AND `identifier =` which signals a
|
|
148
|
+
# new alias declaration).
|
|
149
|
+
rule(:item_continuation) do
|
|
150
|
+
(str("to") | str("before") | str("after") |
|
|
151
|
+
str("not_before") | str("not_after") |
|
|
152
|
+
str("}")).absent? >>
|
|
153
|
+
(identifier >> whitespace? >> str("=")).absent? >>
|
|
154
|
+
item_atom_start
|
|
155
|
+
end
|
|
156
|
+
|
|
157
|
+
rule(:item_atom_start) do
|
|
158
|
+
str('"') | str("'") |
|
|
159
|
+
match(/[A-Za-z_]/)
|
|
160
|
+
end
|
|
161
|
+
|
|
162
|
+
rule(:constraint) do
|
|
163
|
+
(str("before") >> whitespace >> item.as(:before)) |
|
|
164
|
+
(str("after") >> whitespace >> item.as(:after)) |
|
|
165
|
+
(str("not_before") >> whitespace >> item.as(:not_before)) |
|
|
166
|
+
(str("not_after") >> whitespace >> item.as(:not_after))
|
|
167
|
+
end
|
|
168
|
+
|
|
169
|
+
rule(:constraints) do
|
|
170
|
+
(whitespace >> constraint).repeat.as(:constraints)
|
|
171
|
+
end
|
|
172
|
+
end
|
|
173
|
+
end
|
|
174
|
+
end
|
|
175
|
+
end
|
|
176
|
+
end
|
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "parsanol"
|
|
4
|
+
|
|
5
|
+
module Interscript
|
|
6
|
+
module Isc
|
|
7
|
+
module Grammar
|
|
8
|
+
module Concerns
|
|
9
|
+
# Metadata block: identity + provenance + lifecycle of a system.
|
|
10
|
+
module Metadata
|
|
11
|
+
include Parsanol
|
|
12
|
+
|
|
13
|
+
rule(:metadata_block) do
|
|
14
|
+
str("metadata") >> whitespace? >>
|
|
15
|
+
braced(metadata_field.repeat(0)).as(:metadata)
|
|
16
|
+
end
|
|
17
|
+
|
|
18
|
+
rule(:metadata_field) do
|
|
19
|
+
whitespace? >>
|
|
20
|
+
(
|
|
21
|
+
description_field |
|
|
22
|
+
relations_field |
|
|
23
|
+
system_status_field |
|
|
24
|
+
code_status_field |
|
|
25
|
+
specification_field |
|
|
26
|
+
notes_field |
|
|
27
|
+
provenance_field |
|
|
28
|
+
generic_field
|
|
29
|
+
) >> whitespace?
|
|
30
|
+
end
|
|
31
|
+
|
|
32
|
+
rule(:authority_field) do
|
|
33
|
+
str("authority") >> whitespace >> quoted_string.as(:authority)
|
|
34
|
+
end
|
|
35
|
+
|
|
36
|
+
rule(:source_spelling_field) do
|
|
37
|
+
str("source_spelling") >> whitespace >> quoted_string.as(:source_spelling)
|
|
38
|
+
end
|
|
39
|
+
|
|
40
|
+
rule(:target_spelling_field) do
|
|
41
|
+
str("target_spelling") >> whitespace >> quoted_string.as(:target_spelling)
|
|
42
|
+
end
|
|
43
|
+
|
|
44
|
+
rule(:identifying_field) do
|
|
45
|
+
str("identifying") >> whitespace >> quoted_string.as(:identifying)
|
|
46
|
+
end
|
|
47
|
+
|
|
48
|
+
rule(:name_field) do
|
|
49
|
+
str("name") >> whitespace >> quoted_string.as(:name)
|
|
50
|
+
end
|
|
51
|
+
|
|
52
|
+
rule(:specification_field) do
|
|
53
|
+
str("specification") >> whitespace >>
|
|
54
|
+
quoted_string.as(:specification) >>
|
|
55
|
+
(comma >> quoted_string.as(:specification)).repeat
|
|
56
|
+
end
|
|
57
|
+
|
|
58
|
+
rule(:description_field) do
|
|
59
|
+
str("description") >> whitespace? >>
|
|
60
|
+
braced(raw_text.as(:description))
|
|
61
|
+
end
|
|
62
|
+
|
|
63
|
+
rule(:system_status_field) do
|
|
64
|
+
str("system_status") >> whitespace >>
|
|
65
|
+
(str("current") | str("former") | str("inactive")).as(:system_status)
|
|
66
|
+
end
|
|
67
|
+
|
|
68
|
+
rule(:code_status_field) do
|
|
69
|
+
str("code_status") >> whitespace >>
|
|
70
|
+
(str("preferred") | str("proposed") | str("deprecated")).as(:code_status)
|
|
71
|
+
end
|
|
72
|
+
|
|
73
|
+
rule(:relations_field) do
|
|
74
|
+
str("relations") >> whitespace? >>
|
|
75
|
+
braced(relation_block.repeat(0)).as(:relations)
|
|
76
|
+
end
|
|
77
|
+
|
|
78
|
+
rule(:relation_block) do
|
|
79
|
+
whitespace? >>
|
|
80
|
+
relation_type.as(:type) >> whitespace >>
|
|
81
|
+
quoted_string.as(:system) >>
|
|
82
|
+
(whitespace >> str("note") >> whitespace >>
|
|
83
|
+
quoted_string.as(:note)).maybe >>
|
|
84
|
+
whitespace?
|
|
85
|
+
end
|
|
86
|
+
|
|
87
|
+
rule(:relation_type) do
|
|
88
|
+
str("supersedes") | str("superseded_by") | str("based_on") |
|
|
89
|
+
str("basis_for") | str("alias_of") | str("adopted_from") |
|
|
90
|
+
str("related_to")
|
|
91
|
+
end
|
|
92
|
+
|
|
93
|
+
rule(:notes_field) do
|
|
94
|
+
str("notes") >> whitespace? >>
|
|
95
|
+
braced(note_line.repeat(0).as(:notes))
|
|
96
|
+
end
|
|
97
|
+
|
|
98
|
+
rule(:note_line) do
|
|
99
|
+
whitespace? >>
|
|
100
|
+
str("note") >> whitespace? >>
|
|
101
|
+
quoted_string.as(:note) >> whitespace?
|
|
102
|
+
end
|
|
103
|
+
|
|
104
|
+
rule(:provenance_field) do
|
|
105
|
+
str("provenance") >> whitespace? >>
|
|
106
|
+
quoted_string.as(:provenance) >>
|
|
107
|
+
(comma >> quoted_string.as(:provenance)).repeat
|
|
108
|
+
end
|
|
109
|
+
|
|
110
|
+
# Generic field: any identifier followed by a bare value (string,
|
|
111
|
+
# number, or whitespace-delimited tokens up to the next field or
|
|
112
|
+
# close brace). This makes the parser permissive about future
|
|
113
|
+
# field additions; semantic validation happens in DocumentBuilder.
|
|
114
|
+
rule(:generic_field) do
|
|
115
|
+
identifier.as(:field_name) >> inline_whitespace? >>
|
|
116
|
+
(empty_field |
|
|
117
|
+
field_value.as(:field_value) |
|
|
118
|
+
braced(raw_text.as(:field_block)))
|
|
119
|
+
end
|
|
120
|
+
|
|
121
|
+
rule(:field_value) do
|
|
122
|
+
quoted_string |
|
|
123
|
+
(newline.absent? >> (str("}").absent? >> str("{").absent? >> any)).repeat(1).as(:raw)
|
|
124
|
+
end
|
|
125
|
+
|
|
126
|
+
rule(:empty_field) do
|
|
127
|
+
# An identifier with no value (just newline or `}` after). Use
|
|
128
|
+
# lookahead without consuming.
|
|
129
|
+
newline.present? | str("}").present?
|
|
130
|
+
end
|
|
131
|
+
|
|
132
|
+
# Raw text inside `{ ... }` — for description blocks. Consumes any
|
|
133
|
+
# character that isn't an unescaped closing brace. Literal braces
|
|
134
|
+
# inside the body are escaped as `\{` and `\}` by the codemod.
|
|
135
|
+
rule(:raw_text) do
|
|
136
|
+
(str("\\{") | str("\\}") | (str("}").absent? >> any)).repeat
|
|
137
|
+
end
|
|
138
|
+
end
|
|
139
|
+
end
|
|
140
|
+
end
|
|
141
|
+
end
|
|
142
|
+
end
|
|
@@ -0,0 +1,112 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "parsanol"
|
|
4
|
+
|
|
5
|
+
module Interscript
|
|
6
|
+
module Isc
|
|
7
|
+
module Grammar
|
|
8
|
+
module Concerns
|
|
9
|
+
# Lexical primitives shared by every other concern.
|
|
10
|
+
# Mirrors the structure of LutaML LML's Concerns::Primitives.
|
|
11
|
+
module Primitives
|
|
12
|
+
include Parsanol
|
|
13
|
+
|
|
14
|
+
# -- Whitespace and comments
|
|
15
|
+
|
|
16
|
+
rule(:space) { match(/\s/).repeat(1) }
|
|
17
|
+
rule(:space?) { space.maybe }
|
|
18
|
+
|
|
19
|
+
rule(:newline) { str("\n") | str("\r\n") | str("\r") }
|
|
20
|
+
rule(:newlines) { newline.repeat(1) }
|
|
21
|
+
rule(:newlines?) { newlines.maybe }
|
|
22
|
+
|
|
23
|
+
rule(:line_comment) do
|
|
24
|
+
str("#") >> (newline.absent? >> any).repeat
|
|
25
|
+
end
|
|
26
|
+
|
|
27
|
+
rule(:whitespace) do
|
|
28
|
+
(space | line_comment).repeat(1)
|
|
29
|
+
end
|
|
30
|
+
rule(:whitespace?) { whitespace.maybe }
|
|
31
|
+
|
|
32
|
+
# Inline whitespace: spaces and tabs only, NO newlines. Used between
|
|
33
|
+
# a field name and its value to prevent eating the newline that
|
|
34
|
+
# signals an empty value.
|
|
35
|
+
rule(:inline_space) { match(/[ \t]/).repeat(1) }
|
|
36
|
+
rule(:inline_whitespace) do
|
|
37
|
+
(inline_space | line_comment).repeat(1)
|
|
38
|
+
end
|
|
39
|
+
rule(:inline_whitespace?) { inline_whitespace.maybe }
|
|
40
|
+
|
|
41
|
+
# Comma, used in lists. Trailing whitespace allowed.
|
|
42
|
+
rule(:comma) { str(",") >> whitespace? }
|
|
43
|
+
|
|
44
|
+
# Arrow, used in tests.
|
|
45
|
+
rule(:arrow) { whitespace? >> str("->") >> whitespace? }
|
|
46
|
+
|
|
47
|
+
# -- Identifiers
|
|
48
|
+
|
|
49
|
+
rule(:identifier_first) { match(/[a-zA-Z_]/) }
|
|
50
|
+
rule(:identifier_rest) { match(/[a-zA-Z0-9_]/) }
|
|
51
|
+
rule(:identifier) do
|
|
52
|
+
(identifier_first >> identifier_rest.repeat).as(:identifier)
|
|
53
|
+
end
|
|
54
|
+
|
|
55
|
+
# -- String literals
|
|
56
|
+
|
|
57
|
+
rule(:escape_sequence) do
|
|
58
|
+
str("\\") >> (
|
|
59
|
+
str("n").as(:newline) |
|
|
60
|
+
str("r").as(:carriage_return) |
|
|
61
|
+
str("t").as(:tab) |
|
|
62
|
+
str('"').as(:dquote) |
|
|
63
|
+
str("\\").as(:backslash) |
|
|
64
|
+
(str("u") >> match(/[0-9a-fA-F]/).repeat(4, 4).as(:unicode)) |
|
|
65
|
+
# The corpus converter emits \U with 4 hex digits too
|
|
66
|
+
# (gki-bel "\U040E") — accept 4..8, don't be stricter than
|
|
67
|
+
# the corpus.
|
|
68
|
+
(str("U") >> match(/[0-9a-fA-F]/).repeat(4, 8).as(:unicode)) |
|
|
69
|
+
# Converter-corrupted files (un-mar) contain a bare \\U with no
|
|
70
|
+
# hex digits at all — parse it as a literal U so the map loads.
|
|
71
|
+
str("U").as(:u_lone)
|
|
72
|
+
)
|
|
73
|
+
end
|
|
74
|
+
|
|
75
|
+
# Single-quoted strings: no escape interpretation.
|
|
76
|
+
rule(:single_quoted_string) do
|
|
77
|
+
str("'") >>
|
|
78
|
+
(str("'").absent? >> any).repeat.as(:string) >>
|
|
79
|
+
str("'")
|
|
80
|
+
end
|
|
81
|
+
|
|
82
|
+
# Double-quoted strings: \\uXXXX, \\n, etc. are interpreted.
|
|
83
|
+
# Escape-free stretches are captured as ONE run slice — a
|
|
84
|
+
# per-char capture allocates a hash + slice per character
|
|
85
|
+
# (~700k nodes for a large map) and the Transform pattern
|
|
86
|
+
# match grinds on them even under the native parse engine.
|
|
87
|
+
rule(:plain_run) do
|
|
88
|
+
(str("\\") | str('"')).absent? >> any
|
|
89
|
+
end
|
|
90
|
+
rule(:double_quoted_string) do
|
|
91
|
+
str('"') >>
|
|
92
|
+
(escape_sequence | plain_run.repeat(1).as(:run)).repeat.as(:string) >>
|
|
93
|
+
str('"')
|
|
94
|
+
end
|
|
95
|
+
|
|
96
|
+
rule(:quoted_string) do
|
|
97
|
+
double_quoted_string | single_quoted_string
|
|
98
|
+
end
|
|
99
|
+
|
|
100
|
+
# -- Brace-delimited block scaffold
|
|
101
|
+
|
|
102
|
+
# Wrap an inner rule in `{ ... }` with optional surrounding whitespace.
|
|
103
|
+
def braced(inner)
|
|
104
|
+
str("{") >> whitespace? >>
|
|
105
|
+
inner >>
|
|
106
|
+
whitespace? >> str("}")
|
|
107
|
+
end
|
|
108
|
+
end
|
|
109
|
+
end
|
|
110
|
+
end
|
|
111
|
+
end
|
|
112
|
+
end
|
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "parsanol"
|
|
4
|
+
|
|
5
|
+
module Interscript
|
|
6
|
+
module Isc
|
|
7
|
+
module Grammar
|
|
8
|
+
module Concerns
|
|
9
|
+
# Stages: ordered transformation pipelines.
|
|
10
|
+
module Stages
|
|
11
|
+
include Parsanol
|
|
12
|
+
|
|
13
|
+
rule(:stage_block) do
|
|
14
|
+
str("stage") >>
|
|
15
|
+
(str("(") >> identifier.as(:stage_name) >> str(")") |
|
|
16
|
+
whitespace >> identifier.as(:stage_name)) >> whitespace? >>
|
|
17
|
+
braced(stage_item.repeat(0)).as(:stage)
|
|
18
|
+
end
|
|
19
|
+
|
|
20
|
+
rule(:stage_item) do
|
|
21
|
+
whitespace? >>
|
|
22
|
+
(sequence_block | parallel_block | run_rule | separate_directive |
|
|
23
|
+
string_case_directive | compose_directive | rababa_directive |
|
|
24
|
+
bare_rule | comment_item) >>
|
|
25
|
+
whitespace?
|
|
26
|
+
end
|
|
27
|
+
|
|
28
|
+
# Comments and stray identifiers are silently consumed.
|
|
29
|
+
rule(:comment_item) do
|
|
30
|
+
(str("#") >> (str("\n").absent? >> any).repeat).as(:comment) |
|
|
31
|
+
identifier.as(:noop)
|
|
32
|
+
end
|
|
33
|
+
|
|
34
|
+
# Bare rule directly in a stage body (not wrapped in sequence/parallel).
|
|
35
|
+
# The original .imp allows this; treat it as a one-rule sequence.
|
|
36
|
+
rule(:bare_rule) do
|
|
37
|
+
rule.as(:bare_rule)
|
|
38
|
+
end
|
|
39
|
+
|
|
40
|
+
rule(:sequence_block) do
|
|
41
|
+
str("sequence") >> whitespace? >>
|
|
42
|
+
braced(rule_line.repeat(0)).as(:sequence)
|
|
43
|
+
end
|
|
44
|
+
|
|
45
|
+
rule(:parallel_block) do
|
|
46
|
+
str("parallel") >> whitespace? >>
|
|
47
|
+
braced(rule_line.repeat(0)).as(:parallel)
|
|
48
|
+
end
|
|
49
|
+
|
|
50
|
+
rule(:separate_directive) do
|
|
51
|
+
str("separate").as(:separate) >>
|
|
52
|
+
(whitespace >> str("separator") >> whitespace >>
|
|
53
|
+
item_atom.as(:separator)).maybe
|
|
54
|
+
end
|
|
55
|
+
|
|
56
|
+
# `downcase`, `upcase`, `title_case` — string-case directives.
|
|
57
|
+
# title_case takes an optional word_separator kwarg.
|
|
58
|
+
rule(:string_case_directive) do
|
|
59
|
+
(str("downcase") | str("upcase")).as(:case) |
|
|
60
|
+
(str("title_case").as(:case) >>
|
|
61
|
+
(whitespace >> kwarg_list.as(:case_kwargs)).maybe)
|
|
62
|
+
end
|
|
63
|
+
|
|
64
|
+
# `compose` / `decompose` — Unicode normalization directives.
|
|
65
|
+
rule(:compose_directive) do
|
|
66
|
+
str("compose").as(:compose) | str("decompose").as(:compose)
|
|
67
|
+
end
|
|
68
|
+
|
|
69
|
+
# `rababa config: "key"` — invoke the rababa diacritization service.
|
|
70
|
+
# The keyword arguments are passed through to the runtime funcall.
|
|
71
|
+
rule(:rababa_directive) do
|
|
72
|
+
str("rababa").as(:funcall_name) >>
|
|
73
|
+
(whitespace >> kwarg_list.as(:funcall_kwargs)).maybe
|
|
74
|
+
end
|
|
75
|
+
|
|
76
|
+
# `name: "value"` (`,`-separated). Used by rababa and other
|
|
77
|
+
# function-call directives. Each kwarg is captured as a
|
|
78
|
+
# subtree so multiple kwargs collect into a list.
|
|
79
|
+
rule(:kwarg_list) do
|
|
80
|
+
kwarg.as(:kwarg) >> (str(",") >> whitespace? >> kwarg.as(:kwarg)).repeat
|
|
81
|
+
end
|
|
82
|
+
|
|
83
|
+
rule(:kwarg) do
|
|
84
|
+
identifier.as(:kwarg_name) >> whitespace? >>
|
|
85
|
+
str(":") >> whitespace? >>
|
|
86
|
+
quoted_string.as(:kwarg_value)
|
|
87
|
+
end
|
|
88
|
+
|
|
89
|
+
rule(:rule_line) do
|
|
90
|
+
whitespace? >> (rule | comment_item) >> whitespace?
|
|
91
|
+
end
|
|
92
|
+
|
|
93
|
+
# A rule is either compact form (single line) or block form
|
|
94
|
+
# (multi-line). Both produce the same semantic node.
|
|
95
|
+
#
|
|
96
|
+
# In compact form, `from` and `to` are single atoms (no concatenation).
|
|
97
|
+
# This covers the 95% case (`sub "щ" "shch"`). Multi-atom matches
|
|
98
|
+
# require the block form.
|
|
99
|
+
rule(:rule) do
|
|
100
|
+
block_rule | compact_rule
|
|
101
|
+
end
|
|
102
|
+
|
|
103
|
+
rule(:compact_rule) do
|
|
104
|
+
str("sub") >> whitespace >>
|
|
105
|
+
item_atom.as(:from) >> whitespace >>
|
|
106
|
+
item_atom.as(:to) >>
|
|
107
|
+
constraints
|
|
108
|
+
end
|
|
109
|
+
|
|
110
|
+
rule(:block_rule) do
|
|
111
|
+
str("sub") >> whitespace? >>
|
|
112
|
+
str("{") >> whitespace? >>
|
|
113
|
+
str("from") >> whitespace >> item.as(:from) >> whitespace? >>
|
|
114
|
+
str("to") >> whitespace >> item.as(:to) >>
|
|
115
|
+
(whitespace >> constraint).repeat.as(:constraints) >>
|
|
116
|
+
whitespace? >> str("}")
|
|
117
|
+
end
|
|
118
|
+
|
|
119
|
+
rule(:run_rule) do
|
|
120
|
+
str("run") >> whitespace >>
|
|
121
|
+
((str("map.") >> identifier.as(:dep) >>
|
|
122
|
+
str(".stage.") >> identifier.as(:stage)) |
|
|
123
|
+
(str("stage.") >> identifier.as(:stage)).as(:run_stage_only))
|
|
124
|
+
end
|
|
125
|
+
end
|
|
126
|
+
end
|
|
127
|
+
end
|
|
128
|
+
end
|
|
129
|
+
end
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "parsanol"
|
|
4
|
+
|
|
5
|
+
module Interscript
|
|
6
|
+
module Isc
|
|
7
|
+
module Grammar
|
|
8
|
+
module Concerns
|
|
9
|
+
# Top-level system block.
|
|
10
|
+
module System
|
|
11
|
+
include Parsanol
|
|
12
|
+
|
|
13
|
+
# A block-item is any of the top-level constructs that may appear
|
|
14
|
+
# inside `system "<code>" { ... }`.
|
|
15
|
+
rule(:block_item) do
|
|
16
|
+
whitespace? >>
|
|
17
|
+
(metadata_block | aliases_block | tests_block |
|
|
18
|
+
stage_block | dependency_decl) >>
|
|
19
|
+
whitespace?
|
|
20
|
+
end
|
|
21
|
+
|
|
22
|
+
rule(:system_block) do
|
|
23
|
+
str("system") >> whitespace >>
|
|
24
|
+
quoted_string.as(:system_code) >> whitespace? >>
|
|
25
|
+
braced(block_item.repeat(0)).as(:body) >>
|
|
26
|
+
whitespace?
|
|
27
|
+
end
|
|
28
|
+
|
|
29
|
+
# Root rule.
|
|
30
|
+
rule(:isc_source) do
|
|
31
|
+
whitespace? >> system_block.as(:system) >> whitespace?
|
|
32
|
+
end
|
|
33
|
+
end
|
|
34
|
+
end
|
|
35
|
+
end
|
|
36
|
+
end
|
|
37
|
+
end
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "parsanol"
|
|
4
|
+
|
|
5
|
+
module Interscript
|
|
6
|
+
module Isc
|
|
7
|
+
module Grammar
|
|
8
|
+
module Concerns
|
|
9
|
+
# Tests block: normative input/output pairs.
|
|
10
|
+
module Tests
|
|
11
|
+
include Parsanol
|
|
12
|
+
|
|
13
|
+
rule(:tests_block) do
|
|
14
|
+
str("tests") >> whitespace? >>
|
|
15
|
+
braced(test_line.repeat(0)).as(:tests)
|
|
16
|
+
end
|
|
17
|
+
|
|
18
|
+
rule(:test_line) do
|
|
19
|
+
whitespace? >>
|
|
20
|
+
quoted_string.as(:input) >>
|
|
21
|
+
arrow >>
|
|
22
|
+
quoted_string.as(:expected) >>
|
|
23
|
+
(whitespace >> str("note") >> whitespace >>
|
|
24
|
+
quoted_string.as(:note)).maybe >>
|
|
25
|
+
whitespace?
|
|
26
|
+
end
|
|
27
|
+
end
|
|
28
|
+
end
|
|
29
|
+
end
|
|
30
|
+
end
|
|
31
|
+
end
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Interscript
|
|
4
|
+
module Isc
|
|
5
|
+
module Grammar
|
|
6
|
+
module Concerns
|
|
7
|
+
autoload :Primitives, "interscript/isc/grammar/concerns/primitives"
|
|
8
|
+
autoload :Items, "interscript/isc/grammar/concerns/items"
|
|
9
|
+
autoload :Metadata, "interscript/isc/grammar/concerns/metadata"
|
|
10
|
+
autoload :Aliases, "interscript/isc/grammar/concerns/aliases"
|
|
11
|
+
autoload :Tests, "interscript/isc/grammar/concerns/tests"
|
|
12
|
+
autoload :Stages, "interscript/isc/grammar/concerns/stages"
|
|
13
|
+
autoload :Dependencies, "interscript/isc/grammar/concerns/dependencies"
|
|
14
|
+
autoload :System, "interscript/isc/grammar/concerns/system"
|
|
15
|
+
end
|
|
16
|
+
end
|
|
17
|
+
end
|
|
18
|
+
end
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "parsanol"
|
|
4
|
+
|
|
5
|
+
module Interscript
|
|
6
|
+
module Isc
|
|
7
|
+
module Grammar
|
|
8
|
+
# Core grammar: composes every concern into a single Parser ancestor.
|
|
9
|
+
# Mirrors the structure of LutaML LML's Grammar::Core.
|
|
10
|
+
module Core
|
|
11
|
+
include Parsanol
|
|
12
|
+
include Concerns::Primitives
|
|
13
|
+
include Concerns::Items
|
|
14
|
+
include Concerns::Metadata
|
|
15
|
+
include Concerns::Aliases
|
|
16
|
+
include Concerns::Tests
|
|
17
|
+
include Concerns::Stages
|
|
18
|
+
include Concerns::Dependencies
|
|
19
|
+
include Concerns::System
|
|
20
|
+
end
|
|
21
|
+
end
|
|
22
|
+
end
|
|
23
|
+
end
|