expressir 2.4.1 → 2.4.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (35) hide show
  1. checksums.yaml +4 -4
  2. data/TODO.max-perf/01-restore-ci-green.md +36 -0
  3. data/TODO.max-perf/02-streaming-parse-path.md +31 -0
  4. data/TODO.max-perf/03-cli-parallel-opt-in.md +27 -0
  5. data/TODO.max-perf/04-benchmark-harness.md +28 -0
  6. data/TODO.max-perf/05-parallel-fidelity-specs.md +22 -0
  7. data/TODO.max-perf/06-builder-cpu-audit.md +41 -0
  8. data/TODO.max-perf/07-upstream-parsanol-roadmap.md +27 -0
  9. data/TODO.max-perf/08-builder-build-perf.md +45 -0
  10. data/TODO.max-perf/09-grammar-cold-start.md +25 -0
  11. data/TODO.max-perf/10-parser-facade-hygiene.md +23 -0
  12. data/TODO.max-perf/11-ci-green-closeout.md +25 -0
  13. data/TODO.max-perf/12-require-boot-profile.md +25 -0
  14. data/TODO.max-perf/13-key-conversion-specs.md +26 -0
  15. data/TODO.max-perf/14-builder-call-handler-audit.md +28 -0
  16. data/benchmark/srl_benchmark.rb +76 -17
  17. data/expressir.gemspec +1 -1
  18. data/lib/expressir/cli.rb +3 -0
  19. data/lib/expressir/commands/coverage.rb +6 -2
  20. data/lib/expressir/commands/package.rb +4 -1
  21. data/lib/expressir/express/ast_key_converter.rb +114 -0
  22. data/lib/expressir/express/builder.rb +8 -119
  23. data/lib/expressir/express/error.rb +17 -0
  24. data/lib/expressir/express/node_position_index.rb +133 -20
  25. data/lib/expressir/express/parallel_files.rb +229 -0
  26. data/lib/expressir/express/parser.rb +44 -86
  27. data/lib/expressir/express/remark_attacher.rb +34 -20
  28. data/lib/expressir/express/schema_block_scanner.rb +3 -2
  29. data/lib/expressir/express/scope_resolver.rb +34 -5
  30. data/lib/expressir/express.rb +2 -0
  31. data/lib/expressir/model/model_element.rb +6 -1
  32. data/lib/expressir/model/repository.rb +18 -5
  33. data/lib/expressir/version.rb +1 -1
  34. data/lib/expressir.rb +3 -1
  35. metadata +22 -6
@@ -0,0 +1,229 @@
1
+ require "etc"
2
+
3
+ module Expressir
4
+ module Express
5
+ # Fork-based worker pool for parsing many EXPRESS files in parallel.
6
+ # The native parser holds the GVL for the whole parse, so process-level
7
+ # parallelism is the only way to use multiple cores. Files are
8
+ # independent until reference resolution, which stays in the parent.
9
+ #
10
+ # Unlike sequential parsing, the progress block fires in file order
11
+ # only after all files have been parsed.
12
+ class ParallelFiles
13
+ DEFAULT_MAX_PROCESSES = 4
14
+ FRAME_HEADER_BYTES = 4
15
+
16
+ FORK_SUPPORTED = Process.respond_to?(:fork).freeze
17
+
18
+ # Forking is never the default: a library must not spawn processes on
19
+ # behalf of its host (forked children inherit broken thread and lock
20
+ # state, and fork does not exist on all Rubies). Parallelism requires
21
+ # an explicit max_processes > 1 from the caller and a platform that
22
+ # supports fork (e.g. not Windows); otherwise the request degrades
23
+ # to sequential parsing.
24
+ def self.sequential?(files, max_processes)
25
+ !FORK_SUPPORTED || max_processes.nil? || max_processes <= 1 ||
26
+ files.size < 3
27
+ end
28
+
29
+ # @param files [Array<String>] EXPRESS file paths
30
+ # @param max_processes [Integer, nil] worker cap; nil auto-selects
31
+ # @param parse [Proc] callback taking a file path, returning an ExpFile
32
+ # @param strict [Boolean] re-raise every error, including
33
+ # Error::SchemaParseFailure, instead of skipping the file
34
+ # @yield [file, exp_file, error] called in original file order
35
+ # @return [Array<Expressir::Model::ExpFile, nil>] parsed files in order;
36
+ # nil marks a file that failed with Error::SchemaParseFailure
37
+ def self.run(files, parse:, max_processes: nil, strict: false, &block)
38
+ new(files, max_processes, parse, block, strict).run
39
+ end
40
+
41
+ def initialize(files, max_processes, parse, block, strict)
42
+ @files = files
43
+ @parse = parse
44
+ @block = block
45
+ @strict = strict
46
+ @worker_count = [
47
+ files.size - 1,
48
+ max_processes || [Etc.nprocessors, DEFAULT_MAX_PROCESSES].min,
49
+ ].min
50
+ end
51
+
52
+ def run
53
+ job_pipes = Array.new(@worker_count) { IO.pipe }
54
+ result_pipes = Array.new(@worker_count) { IO.pipe }
55
+ pids = spawn_workers(job_pipes, result_pipes)
56
+
57
+ files_results = schedule_jobs(job_pipes, result_pipes)
58
+
59
+ ordered_pass(files_results)
60
+ ensure
61
+ cleanup(job_pipes, result_pipes, pids)
62
+ end
63
+
64
+ private
65
+
66
+ def spawn_workers(job_pipes, result_pipes)
67
+ job_pipes.each_index.map do |i|
68
+ job_r, job_w = job_pipes[i]
69
+ result_r, result_w = result_pipes[i]
70
+ fork do
71
+ job_w.close
72
+ result_r.close
73
+ other_pipes = (job_pipes + result_pipes).flatten -
74
+ [job_r, result_w]
75
+ other_pipes.each { |io| io.close unless io.closed? }
76
+ worker_loop(job_r, result_w)
77
+ end
78
+ end
79
+ end
80
+
81
+ # Assigns one job at a time to whichever worker reports a result, so
82
+ # each file is parsed exactly once and each worker holds at most one
83
+ # in-flight job (keeping result pipes free of interleaved frames).
84
+ def schedule_jobs(job_pipes, result_pipes)
85
+ writers = job_pipes.map(&:last)
86
+ readers = result_pipes.map(&:first)
87
+ files_results = Array.new(@files.size)
88
+ next_job = 0
89
+ busy = {}
90
+
91
+ writers.each_index do |wi|
92
+ break if next_job == @files.size
93
+
94
+ dispatch(writers[wi], next_job)
95
+ busy[wi] = true
96
+ next_job += 1
97
+ end
98
+
99
+ until busy.empty?
100
+ ready, = IO.select(readers.values_at(*busy.keys))
101
+ ready.each do |io|
102
+ wi = readers.index(io)
103
+ payload = Marshal.load(read_frame(io)) # rubocop:disable Security/MarshalLoad
104
+ files_results[payload[:index]] = payload
105
+
106
+ if next_job < @files.size
107
+ dispatch(writers[wi], next_job)
108
+ next_job += 1
109
+ else
110
+ writers[wi].close unless writers[wi].closed?
111
+ busy.delete(wi)
112
+ end
113
+ end
114
+ end
115
+
116
+ files_results
117
+ end
118
+
119
+ def dispatch(writer, job_index)
120
+ write_frame(writer, pack_frame(Marshal.dump([job_index, @files[job_index]])))
121
+ end
122
+
123
+ def worker_loop(job_r, result_w)
124
+ until job_r.eof?
125
+ data = read_frame(job_r)
126
+ index, file = Marshal.load(data) # rubocop:disable Security/MarshalLoad
127
+ begin
128
+ payload = { index: index, exp_file: @parse.call(file) }
129
+ rescue StandardError => e
130
+ payload = { index: index, error: transferable_error(e) }
131
+ end
132
+ write_frame(result_w, pack_frame(Marshal.dump(payload)))
133
+ end
134
+ rescue Errno::EPIPE
135
+ # parent went away; nothing to report to
136
+ exit!(0)
137
+ end
138
+
139
+ # SIGTERM cannot interrupt a worker blocked in the native parser (the
140
+ # GVL is held), so escalate to SIGKILL after a grace period instead of
141
+ # blocking in waitpid forever.
142
+ def cleanup(job_pipes, result_pipes, pids)
143
+ (job_pipes.to_a + result_pipes.to_a).each do |r, w|
144
+ r.close unless r.closed?
145
+ w.close unless w.closed?
146
+ end
147
+ pids.to_a.each { |pid| Process.kill("TERM", pid) if alive?(pid) }
148
+
149
+ deadline = Process.clock_gettime(Process::CLOCK_MONOTONIC) + 2
150
+ pids.to_a.each do |pid|
151
+ while alive?(pid) && Process.clock_gettime(Process::CLOCK_MONOTONIC) < deadline
152
+ Process.waitpid(pid, Process::WNOHANG)
153
+ sleep 0.05
154
+ end
155
+ Process.kill("KILL", pid) if alive?(pid)
156
+ reap(pid)
157
+ end
158
+ end
159
+
160
+ def ordered_pass(files_results)
161
+ files_results.each_with_index do |payload, index|
162
+ error = payload[:error]
163
+ @block&.call(@files[index], payload[:exp_file], error)
164
+ raise error if error && (@strict ||
165
+ !error.is_a?(Error::SchemaParseFailure))
166
+ end
167
+
168
+ files_results.map do |payload|
169
+ payload[:error] ? nil : payload[:exp_file]
170
+ end
171
+ end
172
+
173
+ # Errors carrying native-parser state (e.g. Parsanol::ParseFailed with
174
+ # its cause tree) cannot cross a fork boundary; rebuild them without
175
+ # the untransferable internals, preserving class and message.
176
+ def transferable_error(error)
177
+ Marshal.dump(error)
178
+ error
179
+ rescue StandardError
180
+ if error.is_a?(Error::SchemaParseFailure)
181
+ Error::SchemaParseFailure.new(error.filename,
182
+ StandardError.new(error.message))
183
+ else
184
+ StandardError.new(error.message)
185
+ end
186
+ end
187
+
188
+ def pack_frame(data)
189
+ [data.bytesize].pack("N") + data
190
+ end
191
+
192
+ def write_frame(io, frame)
193
+ io.write(frame)
194
+ end
195
+
196
+ def read_frame(io)
197
+ header = read_exactly(io, FRAME_HEADER_BYTES)
198
+ read_exactly(io, header.unpack1("N"))
199
+ end
200
+
201
+ def read_exactly(io, count)
202
+ data = +""
203
+ while data.bytesize < count
204
+ chunk = io.read(count - data.bytesize)
205
+ unless chunk
206
+ raise Error::ParallelParseError,
207
+ "worker exited before sending its result"
208
+ end
209
+
210
+ data << chunk
211
+ end
212
+ data
213
+ end
214
+
215
+ def alive?(pid)
216
+ Process.kill(0, pid)
217
+ true
218
+ rescue Errno::ESRCH, Errno::EPERM
219
+ false
220
+ end
221
+
222
+ def reap(pid)
223
+ Process.waitpid(pid)
224
+ rescue Errno::ECHILD, Errno::EINVAL
225
+ nil
226
+ end
227
+ end
228
+ end
229
+ end
@@ -100,7 +100,32 @@ module Expressir
100
100
  # @yield [filename, schemas, error] Optional block called for each file
101
101
  # @return [Model::Repository] Repository containing all parsed ExpFiles
102
102
  def self.from_files(files, skip_references: nil, include_source: nil,
103
- root_path: nil, use_native: nil)
103
+ root_path: nil, use_native: nil, max_processes: nil, &progress)
104
+ all_exp_files = if ParallelFiles.sequential?(files, max_processes)
105
+ parse_files_sequentially(
106
+ files, skip_references: skip_references, include_source: include_source,
107
+ root_path: root_path, use_native: use_native
108
+ ) do |file, exp_file, error|
109
+ progress&.call(file, exp_file&.schemas, error)
110
+ end
111
+ else
112
+ ParallelFiles.run(
113
+ files,
114
+ max_processes: max_processes,
115
+ parse: lambda do |file|
116
+ from_file(file, skip_references: true, root_path: root_path,
117
+ use_native: use_native)
118
+ end,
119
+ ) do |file, exp_file, error|
120
+ progress&.call(file, exp_file&.schemas, error)
121
+ end
122
+ end
123
+
124
+ build_repository(all_exp_files, skip_references: skip_references)
125
+ end
126
+
127
+ def self.parse_files_sequentially(files, skip_references: nil,
128
+ include_source: nil, root_path: nil, use_native: nil, &block)
104
129
  all_exp_files = []
105
130
 
106
131
  files.each do |file|
@@ -108,12 +133,19 @@ root_path: nil, use_native: nil)
108
133
  root_path: root_path, use_native: use_native)
109
134
  all_exp_files << exp_file
110
135
 
111
- yield(file, exp_file&.schemas, nil) if block_given?
136
+ yield(file, exp_file, nil) if block
112
137
  rescue StandardError => e
113
- yield(file, nil, e) if block_given?
138
+ # Nil-pad so results align with files by index, exactly like the
139
+ # parallel path does.
140
+ all_exp_files << nil if e.is_a?(Error::SchemaParseFailure)
141
+ yield(file, nil, e) if block
114
142
  raise unless e.is_a?(Error::SchemaParseFailure)
115
143
  end
116
144
 
145
+ all_exp_files
146
+ end
147
+
148
+ def self.build_repository(all_exp_files, skip_references: nil)
117
149
  repository = Model::Repository.new(files: all_exp_files)
118
150
 
119
151
  unless skip_references
@@ -130,15 +162,18 @@ root_path: nil, use_native: nil)
130
162
  # @param skip_references [Boolean] skip resolving references
131
163
  # @param include_source [Boolean] attach original source code to model elements
132
164
  # @param use_native [Boolean] use native parser (default: true when available)
133
- # @param use_streaming [Boolean] use streaming builder for maximum performance
165
+ # @param use_streaming [Boolean] unsupported on current parsanol;
166
+ # passing true raises {Error::StreamingUnsupportedError}. The
167
+ # streaming paths return when parsanol exposes a stable
168
+ # parse_with_builder (see parsanol-ruby#52 and the TODO.max-perf/02
169
+ # notes).
134
170
  # @return [Model::ExpFile] Parsed ExpFile
135
171
  # @raise [Error::SchemaParseFailure] if the content fails to parse
136
172
  def self.from_exp(content, skip_references: nil, include_source: nil,
137
173
  use_native: nil, use_streaming: false)
138
174
  content = strip_bom(content)
139
- if use_streaming && Grammar::Parser.native_available? && defined?(Parsanol::Native.parse_with_builder)
140
- return from_exp_streaming(content, skip_references: skip_references,
141
- include_source: include_source)
175
+ if use_streaming
176
+ raise Error::StreamingUnsupportedError
142
177
  end
143
178
 
144
179
  use_native = Grammar::Parser.native_available? if use_native.nil?
@@ -173,83 +208,6 @@ root_path: nil, use_native: nil)
173
208
  exp_file
174
209
  end
175
210
 
176
- # Parse using streaming builder (construct-by-construct).
177
- # @param content [String] EXPRESS source code
178
- # @param skip_references [Boolean] skip resolving references
179
- # @param include_source [Boolean] attach original source code to model elements
180
- # @return [Model::ExpFile] Parsed ExpFile
181
- # @raise [Error::SchemaParseFailure] if the content fails to parse
182
- def self.from_exp_streaming_builder(content, skip_references: nil,
183
- include_source: nil)
184
- grammar_json = Grammar::Parser.cached_grammar_json
185
- builder = ::Expressir::Express::StreamingBuilder.new(source: content,
186
- include_source: include_source)
187
-
188
- begin
189
- exp_file = Parsanol::Native.parse_with_builder(grammar_json,
190
- content, builder)
191
- rescue StandardError => e
192
- raise Error::SchemaParseFailure.new("(streaming)", e)
193
- end
194
-
195
- exp_file.schemas.each do |schema|
196
- schema.file = nil
197
- schema.file_basename = nil
198
- end
199
-
200
- unless skip_references
201
- Expressir::Benchmark.measure_references do
202
- ResolveReferencesModelVisitor.new.visit(exp_file)
203
- end
204
- end
205
-
206
- exp_file
207
- end
208
-
209
- # Parse each schema separately with fresh arena (memory-bounded).
210
- #
211
- # Splits source into schema blocks via {SchemaBlockScanner} and parses
212
- # each independently. Memory is bounded by the largest schema, not the
213
- # entire file.
214
- #
215
- # @param content [String] EXPRESS source code
216
- # @param skip_references [Boolean] skip resolving references
217
- # @param include_source [Boolean] attach original source code to model elements
218
- # @return [Model::ExpFile] Parsed ExpFile
219
- def self.from_exp_streaming(content, skip_references: nil,
220
- include_source: nil)
221
- grammar_json = Grammar::Parser.cached_schema_grammar_json
222
-
223
- schema_blocks = SchemaBlockScanner.extract_schema_blocks(content)
224
-
225
- schemas = schema_blocks.map do |block|
226
- ast = Parsanol::Native.parse_fresh(grammar_json, block[:source])
227
- schema_model = Builder.build(ast)
228
- schema_model.source = block[:source]
229
- schema_model
230
- rescue StandardError => e
231
- raise Error::SchemaParseFailure.new(
232
- "(schema #{block[:name] || 'unknown'})", e
233
- )
234
- end
235
-
236
- exp_file = Expressir::Model::ExpFile.new
237
- exp_file.schemas = schemas
238
-
239
- exp_file.schemas.each do |schema|
240
- schema.file = nil
241
- schema.file_basename = nil
242
- end
243
-
244
- unless skip_references
245
- Expressir::Benchmark.measure_references do
246
- ResolveReferencesModelVisitor.new.visit(exp_file)
247
- end
248
- end
249
-
250
- exp_file
251
- end
252
-
253
211
  # Transfer file-level untagged remarks that appear before the first
254
212
  # SCHEMA keyword to the first schema's +header+ attribute so they are
255
213
  # accessible via +schema.header+ and through Liquid drops.
@@ -271,8 +229,8 @@ include_source: nil)
271
229
  exp_file.untagged_remarks -= header_remarks
272
230
  end
273
231
  private_class_method :transfer_header_to_schema
274
-
275
- private_class_method :from_exp_streaming
232
+ private_class_method :parse_files_sequentially
233
+ private_class_method :build_repository
276
234
  end
277
235
  end
278
236
  end
@@ -35,6 +35,10 @@ module Expressir
35
35
  # via `child_attributes :foo, :bar, ...`. See TODO.bugs/15.
36
36
  EXPRESSION_CHILDREN = Model::ModelElement.child_attributes_registry
37
37
 
38
+ # `WHERE <label> :` clause headers; the captured label maps to the
39
+ # 1-based line number(s) it appears on.
40
+ WHERE_CLAUSE_PATTERN = /\A\s*WHERE\s+(\w+)\s*:/i
41
+
38
42
  def initialize(source)
39
43
  @source = source
40
44
  @attached_spans = Set.new
@@ -71,6 +75,8 @@ module Expressir
71
75
  @line_map = nil
72
76
  @owner_map = nil
73
77
  @active_scope_map = nil
78
+ @source_lines_for_where_clause = nil
79
+ @where_clause_line_index = nil
74
80
  end
75
81
 
76
82
  private
@@ -239,21 +245,20 @@ module Expressir
239
245
  # shares a line with a node or sits outside any statement-bearing node.
240
246
  def find_body_comment_target(remark)
241
247
  line = remark.line
242
- nodes = @node_index.nodes
243
248
  # An own-line comment shares its line with no node. A node STARTING
244
249
  # here means the remark is an inline tail (code; -- note). The
245
250
  # end-line check is restricted to statements: container end_lines are
246
251
  # child-derived approximations that can collide with comment lines.
247
- return inline_target(remark, nodes) if inline_remark?(remark)
252
+ return inline_target(remark, @node_index.starting_at(line)) if inline_remark?(remark)
248
253
 
249
254
  # A closing keyword on the next code line is decisive: the comment
250
255
  # closes that body. Without this check the comment would instead be
251
256
  # read as leading the next statement of an OUTER region, which is
252
257
  # where it would wrongly render.
253
- closing = closing_region_target(line, nodes)
258
+ closing = closing_region_target(line)
254
259
  return closing if closing.first
255
260
 
256
- enclosing, region, = statement_region_for(line, nodes)
261
+ enclosing, region, = statement_region_for(line)
257
262
  return [nil, nil, nil] unless region
258
263
 
259
264
  following = region
@@ -333,7 +338,7 @@ module Expressir
333
338
  # every span and never reaches statement_region_for. Resolve it from
334
339
  # the keyword that follows: it names both the owner type and the body
335
340
  # being closed.
336
- def closing_region_target(line, nodes)
341
+ def closing_region_target(line)
337
342
  keyword_owner, region, keyword_line = closing_keyword_after(line)
338
343
  return [nil, nil, nil] unless keyword_owner
339
344
 
@@ -345,8 +350,8 @@ module Expressir
345
350
  opener_line = active_opener_line(keyword_line, keyword_owner)
346
351
  return [nil, nil, nil] unless opener_line
347
352
 
348
- owner = nodes.find do |n|
349
- n[:node].is_a?(keyword_owner) && n[:line] == opener_line
353
+ owner = @node_index.starting_at(opener_line).find do |n|
354
+ n[:node].is_a?(keyword_owner)
350
355
  end
351
356
  return [nil, nil, nil] unless owner
352
357
 
@@ -452,18 +457,15 @@ module Expressir
452
457
  [nil, nil, nil]
453
458
  end
454
459
 
455
- def statement_region_for(line, nodes)
456
- candidates = nodes.select do |n|
460
+ def statement_region_for(line, nodes = nil)
461
+ candidates = (nodes || @node_index.spanning(line)).select do |n|
457
462
  n[:line] && n[:end_line] && n[:line] <= line && n[:end_line] >= line &&
458
463
  (n[:node].is_a?(Model::Statement) || function_rule_procedure?(n[:node]))
459
464
  end
460
465
  enclosing = innermost_candidate(candidates)
461
466
  return [nil, nil, nil] unless enclosing
462
467
 
463
- children = nodes.select do |n|
464
- n[:owner].equal?(enclosing[:node]) &&
465
- STATEMENT_REGIONS.include?(n[:collection]) && n[:line]
466
- end
468
+ children = @node_index.children_in(enclosing[:node], STATEMENT_REGIONS)
467
469
  return [enclosing, nil, nil] if children.empty?
468
470
 
469
471
  preceding = children.select { |n| n[:line] < line }.max_by { |n| n[:position] }
@@ -658,27 +660,39 @@ module Expressir
658
660
  where_rules = get_collection(scope, :where_rules)
659
661
  return nil unless where_rules&.any?
660
662
 
661
- lines = source_lines_for_where_clause
663
+ where_clause_lines = where_clause_line_index
662
664
 
663
665
  where_rules.each do |wr|
664
666
  next unless wr.id
665
667
 
666
- lines.each_with_index do |line, idx|
667
- line_num = idx + 1
668
+ where_clause_lines.fetch(wr.id, []).each do |line_num|
668
669
  next unless line_num < remark_line
669
670
 
670
- if (line =~ /^\s*WHERE\s+#{Regexp.escape(wr.id)}\s*:/i) && remark_line.between?(line_num, line_num + 5)
671
- return create_remark_item(wr, tag)
672
- end
671
+ return create_remark_item(wr, tag) if remark_line.between?(line_num, line_num + 5)
673
672
  end
674
673
  end
675
674
 
676
675
  nil
677
676
  end
678
677
 
678
+ # Single scan over the source lines mapping each `WHERE <id>:` label to
679
+ # its 1-based line number, so per-remark lookups stop re-testing every
680
+ # line against every WHERE rule's regex.
681
+ def where_clause_line_index
682
+ @where_clause_line_index ||= begin
683
+ index = Hash.new { |h, k| h[k] = [] }
684
+ source_lines_for_where_clause.each_with_index do |line, idx|
685
+ if (match = line.match(WHERE_CLAUSE_PATTERN))
686
+ index[match[1]] << (idx + 1)
687
+ end
688
+ end
689
+ index
690
+ end
691
+ end
692
+
679
693
  def source_lines_for_where_clause
680
694
  # @source is set for the duration of `attach`; freed at the end.
681
- @source.lines
695
+ @source_lines_for_where_clause ||= @source.lines
682
696
  end
683
697
 
684
698
  def find_node_in_statement(stmt, tag)
@@ -6,8 +6,9 @@ module Expressir
6
6
  #
7
7
  # A hand-rolled state machine that skips comments (`(* ... *)`) and
8
8
  # string literals, tracks SCHEMA/END_SCHEMA depth, and returns one
9
- # block per schema declaration. Used by `Parser.from_exp_streaming`
10
- # to parse each schema independently with a fresh memory arena.
9
+ # block per schema declaration. Retained for the future streaming
10
+ # parse path (see TODO.max-perf/02) that parses each schema
11
+ # independently with a fresh memory arena.
11
12
  #
12
13
  # Extracted from the outer Parser class (TODO.bugs/09) so block-
13
14
  # scanning logic lives in one focused module behind one interface.
@@ -50,8 +50,13 @@ module Expressir
50
50
  @model = model
51
51
  @nodes_with_positions = nodes_with_positions
52
52
  @scope_map = nil
53
+ @source_lines = nil
54
+ @position_buckets = nil
53
55
  end
54
56
 
57
+ # Line-band width for the position fallback index.
58
+ BUCKET_LINES = 1024
59
+
55
60
  # Returns the innermost Model::ScopeContainer whose source span contains
56
61
  # the given 1-based remark line, or nil if none is found.
57
62
  def containing_scope_for(remark_line)
@@ -69,7 +74,7 @@ module Expressir
69
74
  type_state = { line: nil, name: nil }
70
75
  rule_state = { line: nil, name: nil }
71
76
 
72
- @source.lines.each_with_index do |line, idx|
77
+ source_lines.each_with_index do |line, idx|
73
78
  line_num = idx + 1
74
79
 
75
80
  case line
@@ -131,8 +136,15 @@ module Expressir
131
136
  @scope_map ||= build_scope_map
132
137
  end
133
138
 
139
+ # Source split into lines, computed once per resolver: both the
140
+ # scope-map build and find_by_source_text need it, and re-splitting the
141
+ # whole source per lookup dominates remark attachment on large files.
142
+ def source_lines
143
+ @source_lines ||= @source.lines
144
+ end
145
+
134
146
  def build_scope_map
135
- lines = @source.lines
147
+ lines = source_lines
136
148
  map = {}
137
149
  return map if lines.empty?
138
150
 
@@ -160,10 +172,27 @@ module Expressir
160
172
 
161
173
  # --- Strategy 2: position-based fallback against the node index ---
162
174
 
175
+ # Nodes bucketed by 1024-line bands: a node spanning [line, end_line]
176
+ # is registered in every band it overlaps, so a remark-line lookup only
177
+ # scans nodes that can possibly contain it. Bucket order preserves the
178
+ # original index order, so select+reverse_each semantics are unchanged.
179
+ def position_buckets
180
+ @position_buckets ||= begin
181
+ buckets = Hash.new { |h, k| h[k] = [] }
182
+ @nodes_with_positions.each do |n|
183
+ next unless n[:line] && n[:end_line]
184
+
185
+ ((n[:line] / BUCKET_LINES)..(n[:end_line] / BUCKET_LINES)).each do |b|
186
+ buckets[b] << n
187
+ end
188
+ end
189
+ buckets
190
+ end
191
+ end
192
+
163
193
  def find_by_position(remark_line)
164
- containing = @nodes_with_positions.select do |n|
165
- n[:line] && n[:end_line] &&
166
- remark_line >= n[:line] && remark_line <= n[:end_line] &&
194
+ containing = position_buckets[remark_line / BUCKET_LINES].select do |n|
195
+ remark_line.between?(n[:line], n[:end_line]) &&
167
196
  !n[:node].is_a?(Model::Repository) && !n[:node].is_a?(Model::Cache)
168
197
  end
169
198
 
@@ -2,6 +2,7 @@
2
2
 
3
3
  module Expressir
4
4
  module Express
5
+ autoload :AstKeyConverter, "#{__dir__}/express/ast_key_converter"
5
6
  autoload :Builder, "#{__dir__}/express/builder"
6
7
  autoload :BuilderContext, "#{__dir__}/express/builder_context"
7
8
  autoload :BuilderRegistry, "#{__dir__}/express/builder_registry"
@@ -17,6 +18,7 @@ module Expressir
17
18
  autoload :LineMap, "#{__dir__}/express/line_map"
18
19
  autoload :ModelVisitor, "#{__dir__}/express/model_visitor"
19
20
  autoload :NodePositionIndex, "#{__dir__}/express/node_position_index"
21
+ autoload :ParallelFiles, "#{__dir__}/express/parallel_files"
20
22
  autoload :Parser, "#{__dir__}/express/parser"
21
23
  autoload :PrettyFormatter, "#{__dir__}/express/pretty_formatter"
22
24
  autoload :RemarkAttacher, "#{__dir__}/express/remark_attacher"
@@ -155,7 +155,12 @@ module Expressir
155
155
  end
156
156
 
157
157
  def source
158
- Expressir::Express::SourceFormatter.format(self)
158
+ # Formatting is not free and callers probe `source` repeatedly while
159
+ # walking the tree (position index, remark attachment), so the text is
160
+ # computed once per node. Deliberately NOT @source: that ivar holds
161
+ # the raw source span lutaml-model's `source=` stores when
162
+ # include_source is on, and readers must keep seeing formatted text.
163
+ @formatted_source ||= Expressir::Express::SourceFormatter.format(self) # rubocop:disable Naming/MemoizedInstanceVariableName
159
164
  end
160
165
 
161
166
  # @param [Hash] options
@@ -291,14 +291,27 @@ reference_index: nil)
291
291
  # Build repository from list of schema files
292
292
  # @param file_paths [Array<String>] Schema file paths
293
293
  # @param base_dir [String, nil] Base directory for path resolution
294
+ # @param max_processes [Integer, nil] parse in a fork worker pool when
295
+ # set above 1 on a fork-capable platform (advisory; degrades to
296
+ # sequential otherwise). Unlike Express::Parser.from_files, a file
297
+ # that fails to parse always raises.
294
298
  # @return [Repository] Built repository with all schemas
295
- def self.from_files(file_paths, base_dir: nil)
299
+ def self.from_files(file_paths, base_dir: nil, max_processes: nil)
296
300
  repo = new(base_dir: base_dir)
297
301
 
298
- file_paths.each do |path|
299
- parsed = Expressir::Express::Parser.from_file(path)
300
- next unless parsed
301
-
302
+ pool = Expressir::Express::ParallelFiles
303
+ parsed_files = if pool.sequential?(file_paths, max_processes)
304
+ file_paths.map do |path|
305
+ Expressir::Express::Parser.from_file(path)
306
+ end
307
+ else
308
+ pool.run(file_paths,
309
+ parse: ->(path) { Expressir::Express::Parser.from_file(path) },
310
+ max_processes: max_processes,
311
+ strict: true)
312
+ end
313
+
314
+ parsed_files.each do |parsed|
302
315
  repo.files << parsed if parsed.is_a?(ExpFile)
303
316
  end
304
317