carray-jit 0.1.2 → 0.1.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +771 -3
- data/README.md +7 -6
- data/carray-jit.gemspec +1 -3
- data/docs/00_Introduction.md +4 -3
- data/docs/01_GettingStarted.md +1 -1
- data/docs/02_KernelShapes.md +93 -14
- data/docs/03_SupportedFeatures.md +582 -26
- data/docs/04_Compiling.md +33 -6
- data/docs/05_DesignNotes.md +3 -3
- data/docs/06_Cheatsheet.md +198 -5
- data/docs/07_StepByStep.ja.md +534 -0
- data/docs/07_StepByStep.md +535 -0
- data/examples/README.md +12 -0
- data/examples/applications/alarm.rb +121 -0
- data/examples/applications/collatz.rb +105 -0
- data/examples/applications/cubic_spline.rb +331 -0
- data/examples/applications/dithering.rb +144 -0
- data/examples/applications/group_stats.rb +115 -0
- data/examples/applications/lookup.rb +126 -0
- data/examples/applications/median_filter.rb +153 -0
- data/examples/applications/parcel_ascent.rb +220 -0
- data/examples/applications/point_in_polygon.rb +111 -0
- data/examples/applications/random_walk.rb +98 -0
- data/examples/applications/van_der_pol.rb +186 -0
- data/examples/applications/wet_bulb.rb +140 -0
- data/examples/features/10_complex.rb +14 -4
- data/examples/features/15_loops.rb +7 -1
- data/lib/carray/jit/access.rb +14 -0
- data/lib/carray/jit/analyzer.rb +2077 -136
- data/lib/carray/jit/block_reader.rb +37 -6
- data/lib/carray/jit/c_function.rb +613 -76
- data/lib/carray/jit/c_generator.rb +1595 -156
- data/lib/carray/jit/call.rb +68 -0
- data/lib/carray/jit/compiler.rb +75 -11
- data/lib/carray/jit/kernel.rb +369 -32
- data/lib/carray/jit/node.rb +359 -9
- data/lib/carray/jit/sorting_networks.rb +182 -0
- data/lib/carray/jit/type_assignment.rb +371 -34
- data/lib/carray/jit/version.rb +1 -1
- data/lib/carray/jit.rb +560 -64
- metadata +22 -8
- data/ext/carray_jit_access/carray_jit_access.c +0 -460
- data/ext/carray_jit_access/extconf.rb +0 -8
data/lib/carray/jit/kernel.rb
CHANGED
|
@@ -19,7 +19,13 @@ class CArray
|
|
|
19
19
|
# @return [Boolean] whether this call invoked the compiler, rather
|
|
20
20
|
# than reusing a cached object.
|
|
21
21
|
attr_reader :source, :c_source, :arrays, :storage_types, :reals,
|
|
22
|
-
:integers, :complexes, :
|
|
22
|
+
:integers, :complexes, :unsigned_integers,
|
|
23
|
+
:rank, :index_names, :written_arrays,
|
|
24
|
+
# The arrays handed to a C function whole, by address,
|
|
25
|
+
# rather than walked a cell at a time. A caller that lines
|
|
26
|
+
# operands up has to leave these alone: their shape is the
|
|
27
|
+
# C function's business, not the expression's.
|
|
28
|
+
:address_arrays,
|
|
23
29
|
:compiled, :masked, :directions, :contracted_names,
|
|
24
30
|
:index_axes,
|
|
25
31
|
# How far a stencil's windows reach on each axis, as
|
|
@@ -38,6 +44,7 @@ class CArray
|
|
|
38
44
|
@reals = generator.reals
|
|
39
45
|
@integers = generator.integers
|
|
40
46
|
@complexes = generator.complexes
|
|
47
|
+
@unsigned_integers = generator.unsigned_integers
|
|
41
48
|
# Per array, the axes read at an index only the running kernel knows,
|
|
42
49
|
# and the extents it checks them against.
|
|
43
50
|
@extent_slots = generator.extent_slots
|
|
@@ -48,6 +55,9 @@ class CArray
|
|
|
48
55
|
@written_arrays = analyzer.written_arrays
|
|
49
56
|
@array_ranks = analyzer.array_ranks
|
|
50
57
|
@inner_ranges = analyzer.inner_ranges
|
|
58
|
+
# An index the block wrote twice has two identifiers here; the
|
|
59
|
+
# messages speak the name it was written with.
|
|
60
|
+
@index_sources = analyzer.index_sources
|
|
51
61
|
@axis_uses = @arrays.to_h { |array|
|
|
52
62
|
[array, (0...array_rank(array)).map { |axis| analyzer.axis_use(array, axis) }]
|
|
53
63
|
}
|
|
@@ -61,6 +71,17 @@ class CArray
|
|
|
61
71
|
end
|
|
62
72
|
end
|
|
63
73
|
|
|
74
|
+
# Which captured scalars decide where the kernel reaches, rather than
|
|
75
|
+
# what it computes. A pinned subscript, an offset and an inner loop's
|
|
76
|
+
# range are worked out here, in Ruby, before the first cell; every
|
|
77
|
+
# other captured number is packed into a buffer and read by the C. So
|
|
78
|
+
# these are the ones a prepared call has to be keyed on -- and a
|
|
79
|
+
# coefficient or a clock that changes every step is not one of them,
|
|
80
|
+
# which is what keeps a time-stepping caller on the prepared path.
|
|
81
|
+
@shape_scalar_names = shape_scalar_names
|
|
82
|
+
# Calls prepared earlier, most recently used first.
|
|
83
|
+
@plans = []
|
|
84
|
+
|
|
64
85
|
@masked = generator.masked
|
|
65
86
|
@window_reach = analyzer.window_reach
|
|
66
87
|
@window_reaches = analyzer.window_reaches
|
|
@@ -68,6 +89,15 @@ class CArray
|
|
|
68
89
|
# writes into the error slot. The message does not travel: C has
|
|
69
90
|
# nothing to carry it in, and it was known when this was compiled.
|
|
70
91
|
@raise_messages = generator.raise_messages
|
|
92
|
+
# The arrays this kernel allocates at its entry, as
|
|
93
|
+
# [name, storage, shape] -- so that a shape it could not use, or an
|
|
94
|
+
# allocation that failed, can be named with the lengths it came to on
|
|
95
|
+
# this call. The shapes are worked out here as they are there: from
|
|
96
|
+
# the same expressions over the same captured integers.
|
|
97
|
+
@heap_arrays = analyzer.local_array_declarations.filter_map { |entry|
|
|
98
|
+
_scope, name, storage, shape, heap = entry
|
|
99
|
+
[name, storage, shape] if heap
|
|
100
|
+
}
|
|
71
101
|
# Kept so the sweep entry point can be built from it if one is ever
|
|
72
102
|
# asked for. Compiling it eagerly would pay for a road most kernels
|
|
73
103
|
# never take.
|
|
@@ -109,16 +139,15 @@ class CArray
|
|
|
109
139
|
def call (array_values, scalar_values, bounds, c_function_values = {},
|
|
110
140
|
border: false)
|
|
111
141
|
arrays = @arrays.map { |name| array_values.fetch(name) }
|
|
112
|
-
|
|
113
|
-
verify(arrays, bounds, ranges, scalar_values, reach: !border)
|
|
142
|
+
plan = prepared_call(arrays, array_values, bounds, scalar_values, border)
|
|
114
143
|
|
|
115
|
-
writable =
|
|
144
|
+
writable = plan[:writable]
|
|
116
145
|
error = [0].pack("l")
|
|
117
|
-
packed_bounds = bounds
|
|
146
|
+
packed_bounds = plan[:bounds]
|
|
118
147
|
reals = packed_reals(scalar_values)
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
148
|
+
# The captured integers change from call to call and are packed here;
|
|
149
|
+
# the extents behind them are the arrays' own and travel in the plan.
|
|
150
|
+
integers = packed_scalar_integers(scalar_values) + plan[:extents]
|
|
122
151
|
functions = @c_function_names.map { |name|
|
|
123
152
|
c_function_values.fetch(name).pointer.to_i
|
|
124
153
|
}.pack("Q*")
|
|
@@ -136,14 +165,12 @@ class CArray
|
|
|
136
165
|
}
|
|
137
166
|
}.pack("Q*")
|
|
138
167
|
|
|
139
|
-
box =
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
basis[:mask_strides] || Array.new(@rank, 0)
|
|
146
|
-
}.pack("q*")
|
|
168
|
+
box = plan[:box]
|
|
169
|
+
# Opened for a kernel rather than for a reader: the four buffers come
|
|
170
|
+
# back packed, instead of a hash and two arrays per operand that this
|
|
171
|
+
# would pack back into bytes and throw away.
|
|
172
|
+
Access.open(arrays, writable, box[0], box[1], true) do
|
|
173
|
+
|pointers, strides, mask_pointers, mask_strides|
|
|
147
174
|
entry = border ? border_function : @function
|
|
148
175
|
entry.call(buffer(pointers), buffer(strides), buffer(packed_bounds),
|
|
149
176
|
buffer(reals), buffer(integers), buffer(functions),
|
|
@@ -157,9 +184,17 @@ class CArray
|
|
|
157
184
|
array[] = buffer
|
|
158
185
|
end
|
|
159
186
|
|
|
160
|
-
report(error)
|
|
187
|
+
report(error, scalar_values)
|
|
161
188
|
end
|
|
162
189
|
|
|
190
|
+
# How many prepared calls one kernel keeps. A stencil asks under its
|
|
191
|
+
# interior's bounds and one pair per axis of frame, so a two-dimensional
|
|
192
|
+
# one needs five before it repeats; a caller alternating between two
|
|
193
|
+
# model states needs two of whatever it was already using. Past this
|
|
194
|
+
# the oldest goes and the call prepares itself again, which is what
|
|
195
|
+
# every call did before.
|
|
196
|
+
PLAN_LIMIT = 8
|
|
197
|
+
|
|
163
198
|
# @private
|
|
164
199
|
SLAB_NAME = "carray_jit_slab"
|
|
165
200
|
|
|
@@ -178,10 +213,7 @@ class CArray
|
|
|
178
213
|
|
|
179
214
|
error = [0].pack("l")
|
|
180
215
|
reals = packed_reals(scalar_values)
|
|
181
|
-
integers = (
|
|
182
|
-
@extent_slots.map { |name, axis|
|
|
183
|
-
array_values.fetch(name).dim[axis]
|
|
184
|
-
}).pack("q*")
|
|
216
|
+
integers = packed_integers(scalar_values, array_values)
|
|
185
217
|
functions = @c_function_names.map { |name|
|
|
186
218
|
c_function_values.fetch(name).pointer.to_i
|
|
187
219
|
}.pack("Q*")
|
|
@@ -211,7 +243,7 @@ class CArray
|
|
|
211
243
|
array[] = item
|
|
212
244
|
end
|
|
213
245
|
|
|
214
|
-
report(error)
|
|
246
|
+
report(error, scalar_values)
|
|
215
247
|
end
|
|
216
248
|
|
|
217
249
|
# The wrapper CArray's sweep calls, for the same reason #c_source is
|
|
@@ -230,16 +262,138 @@ class CArray
|
|
|
230
262
|
|
|
231
263
|
private
|
|
232
264
|
|
|
265
|
+
# The parts of a call that the operands' shapes, the bounds and the
|
|
266
|
+
# kernel's own scalars decide, rather than the data or the views.
|
|
267
|
+
#
|
|
268
|
+
# Everything here answers the same for two calls whose key agrees, so it
|
|
269
|
+
# is worked out once and kept: the bounds check, the box the xfer tier
|
|
270
|
+
# transfers, the extents the kernel is handed, and which operands it
|
|
271
|
+
# writes. What is deliberately *not* here is anything read off a view:
|
|
272
|
+
# two arrays of one shape and one type can be laid out differently, and
|
|
273
|
+
# a stride read from the one is not a stride into the other. Those come
|
|
274
|
+
# from Access.open on every call, as they always did.
|
|
275
|
+
#
|
|
276
|
+
# Nor is the refusal of a masked array handed whole to a C function
|
|
277
|
+
# (#address_buffer): a mask is runtime state that leaves the shape and
|
|
278
|
+
# the type alone, so no key here would notice one appearing. It is
|
|
279
|
+
# asked on every call, which costs a predicate and keeps the refusal.
|
|
280
|
+
def prepared_call (arrays, array_values, bounds, scalars, border)
|
|
281
|
+
key = plan_key(arrays, bounds, scalars, border)
|
|
282
|
+
if key
|
|
283
|
+
found = @plans.assoc(key)
|
|
284
|
+
if found
|
|
285
|
+
unless @plans.first.equal?(found)
|
|
286
|
+
@plans.unshift(@plans.delete(found))
|
|
287
|
+
end
|
|
288
|
+
return found[1]
|
|
289
|
+
end
|
|
290
|
+
end
|
|
291
|
+
ranges = index_ranges(bounds, scalars)
|
|
292
|
+
verify(arrays, bounds, ranges, scalars, reach: !border)
|
|
293
|
+
plan = {
|
|
294
|
+
:writable => @arrays.map { |name| @written_arrays.include?(name) },
|
|
295
|
+
:bounds => bounds.flatten.pack("q*"),
|
|
296
|
+
:extents => packed_extents(array_values),
|
|
297
|
+
:box => region_box(ranges, scalars, array_values),
|
|
298
|
+
}
|
|
299
|
+
freeze_throughout(plan)
|
|
300
|
+
if key
|
|
301
|
+
@plans.unshift([key, plan])
|
|
302
|
+
@plans.pop if @plans.size > PLAN_LIMIT
|
|
303
|
+
end
|
|
304
|
+
plan
|
|
305
|
+
end
|
|
306
|
+
|
|
307
|
+
# What two calls have to agree on for one to be prepared like the other.
|
|
308
|
+
#
|
|
309
|
+
# Shapes and types, because that is all `verify` and `region_box` and
|
|
310
|
+
# the extents read of an array; the bounds and the border, because they
|
|
311
|
+
# are what the check is against; and the scalars that decide where the
|
|
312
|
+
# kernel reaches, but not the ones it merely computes with.
|
|
313
|
+
#
|
|
314
|
+
# No identity: two arrays of one shape and type prepare the same way, so
|
|
315
|
+
# keying on which object it was would only make a caller that rebuilds
|
|
316
|
+
# its fields every step miss every time -- and would have this kernel,
|
|
317
|
+
# which lives as long as the process, hold their memory.
|
|
318
|
+
#
|
|
319
|
+
# nil where an operand is not a CArray at all, so that the call falls
|
|
320
|
+
# through to `verify` and is refused by name rather than by whatever
|
|
321
|
+
# the next method call happens to raise.
|
|
322
|
+
def plan_key (arrays, bounds, scalars, border)
|
|
323
|
+
shapes = []
|
|
324
|
+
arrays.each do |array|
|
|
325
|
+
return nil unless array.is_a?(CArray)
|
|
326
|
+
shapes << array.data_type << array.dim
|
|
327
|
+
end
|
|
328
|
+
[shapes, bounds, border,
|
|
329
|
+
@shape_scalar_names.map { |name| scalars[name] }]
|
|
330
|
+
end
|
|
331
|
+
|
|
332
|
+
# A kernel is shared, and Fiddle lets go of the GVL for the call, so two
|
|
333
|
+
# threads can be reading one plan while a third prepares another. That
|
|
334
|
+
# is sound only while nothing in a plan is written after it is published
|
|
335
|
+
# -- and a box is arrays inside an array, so freezing the top of it
|
|
336
|
+
# would leave the parts a later edit could still reach.
|
|
337
|
+
def freeze_throughout (value)
|
|
338
|
+
case value
|
|
339
|
+
when Hash then value.each_value { |item| freeze_throughout(item) }
|
|
340
|
+
when Array then value.each { |item| freeze_throughout(item) }
|
|
341
|
+
end
|
|
342
|
+
value.freeze
|
|
343
|
+
end
|
|
344
|
+
|
|
345
|
+
# The captured scalars a pinned subscript, an offset or an inner loop's
|
|
346
|
+
# range is worked out from.
|
|
347
|
+
def shape_scalar_names
|
|
348
|
+
names = []
|
|
349
|
+
@axis_uses.each_value do |uses|
|
|
350
|
+
uses.each do |walkers, pinned|
|
|
351
|
+
pinned.each { |node| collect_captures(node, names) }
|
|
352
|
+
walkers.each { |walker| walker[1].each { |node|
|
|
353
|
+
collect_captures(node, names)
|
|
354
|
+
} }
|
|
355
|
+
end
|
|
356
|
+
end
|
|
357
|
+
@inner_ranges.each_value do |from, to, _step|
|
|
358
|
+
collect_captures(from, names)
|
|
359
|
+
collect_captures(to, names)
|
|
360
|
+
end
|
|
361
|
+
names.uniq.freeze
|
|
362
|
+
end
|
|
363
|
+
|
|
364
|
+
def collect_captures (node, names)
|
|
365
|
+
return names unless node.is_a?(Node)
|
|
366
|
+
names << node.name if node.is_a?(CaptureRead)
|
|
367
|
+
node.children.each { |child| collect_captures(child, names) }
|
|
368
|
+
names
|
|
369
|
+
end
|
|
370
|
+
|
|
233
371
|
# What the cell that stopped said, raised now that Ruby has control
|
|
234
372
|
# back. C cannot raise, so a failure is a code in a slot and a loop
|
|
235
373
|
# that stops; this is where it becomes the exception the same body run
|
|
236
374
|
# in Ruby would have raised.
|
|
237
|
-
def report (error)
|
|
375
|
+
def report (error, scalars = {})
|
|
238
376
|
code = error.unpack1("l")
|
|
377
|
+
if (failure = CGenerator::FIXED_FAILURES[code])
|
|
378
|
+
raise failure[0], failure[1]
|
|
379
|
+
end
|
|
239
380
|
case code
|
|
240
381
|
when 0 then nil
|
|
382
|
+
when CGenerator::SHAPE_CODE then raise_a_bad_shape(scalars)
|
|
383
|
+
when CGenerator::MEMORY_CODE then raise_out_of_memory(scalars)
|
|
241
384
|
when 1 then raise ZeroDivisionError, "divided by 0"
|
|
242
385
|
when 2 then raise IndexError, "index out of range"
|
|
386
|
+
when 3 then raise ArgumentError,
|
|
387
|
+
"min argument must be less than or equal to " \
|
|
388
|
+
"max argument"
|
|
389
|
+
# Ruby's own wording names the value it could not compare, and the
|
|
390
|
+
# slot carries a code rather than a number, so the reason is named
|
|
391
|
+
# instead. The class and the failure are Ruby's.
|
|
392
|
+
when 4 then raise ArgumentError,
|
|
393
|
+
"comparison with a NaN failed, so `clamp` has " \
|
|
394
|
+
"no answer"
|
|
395
|
+
when 5 then raise Math::DomainError,
|
|
396
|
+
"Numerical argument is out of domain - gamma"
|
|
243
397
|
else
|
|
244
398
|
message = @raise_messages[code]
|
|
245
399
|
# A code with no message behind it is this compiler's bug, not the
|
|
@@ -251,6 +405,53 @@ class CArray
|
|
|
251
405
|
end
|
|
252
406
|
end
|
|
253
407
|
|
|
408
|
+
# A length the kernel worked out that is no length: zero cells, a
|
|
409
|
+
# negative count, or so many that the bytes would not be a `size_t`.
|
|
410
|
+
# The kernel reports which of its arrays it was by reporting at all --
|
|
411
|
+
# it stops before the first cell -- and the lengths are worked out
|
|
412
|
+
# again here, where there is a message to put them in.
|
|
413
|
+
def raise_a_bad_shape (scalars)
|
|
414
|
+
offending = @heap_arrays.filter_map { |name, storage, shape|
|
|
415
|
+
lengths = shape_lengths(shape, scalars)
|
|
416
|
+
next unless lengths&.any? { |length| length < 1 }
|
|
417
|
+
"`#{name}` (#{storage}) asks for " \
|
|
418
|
+
"#{lengths.join(' x ')} #{lengths.size == 1 ? 'cells' : 'cells per axis'}"
|
|
419
|
+
}
|
|
420
|
+
raise ArgumentError,
|
|
421
|
+
"a local array holds at least one cell: " \
|
|
422
|
+
"#{offending.join('; ')}" unless offending.empty?
|
|
423
|
+
raise ArgumentError,
|
|
424
|
+
"a local array of this kernel asks for more cells than its " \
|
|
425
|
+
"bytes could be counted in"
|
|
426
|
+
end
|
|
427
|
+
|
|
428
|
+
def raise_out_of_memory (scalars)
|
|
429
|
+
asked = @heap_arrays.map { |name, storage, shape|
|
|
430
|
+
lengths = shape_lengths(shape, scalars)
|
|
431
|
+
cells = lengths&.inject(1, :*)
|
|
432
|
+
bytes = cells && cells * LOCAL_ARRAY_STORAGE_BYTES.fetch(storage, 1)
|
|
433
|
+
"`#{name}` (#{storage}, #{lengths ? lengths.join(' x ') : 'shape'}" +
|
|
434
|
+
(bytes ? ", #{bytes} bytes" : "") + ")"
|
|
435
|
+
}
|
|
436
|
+
raise NoMemoryError,
|
|
437
|
+
"a kernel could not allocate the local arrays it works in: " \
|
|
438
|
+
"#{asked.join(', ')}"
|
|
439
|
+
end
|
|
440
|
+
|
|
441
|
+
# @private
|
|
442
|
+
LOCAL_ARRAY_STORAGE_BYTES = Analyzer::LOCAL_ARRAY_STORAGE_BYTES
|
|
443
|
+
|
|
444
|
+
# The shape as numbers, or nil where one of its axes cannot be worked
|
|
445
|
+
# out here -- which is not a failure worth a second exception, the one
|
|
446
|
+
# being raised having happened already.
|
|
447
|
+
def shape_lengths (shape, scalars)
|
|
448
|
+
shape.map { |extent|
|
|
449
|
+
extent.is_a?(Integer) ? extent : evaluate(extent.node, scalars)
|
|
450
|
+
}
|
|
451
|
+
rescue StandardError, Unsupported
|
|
452
|
+
nil
|
|
453
|
+
end
|
|
454
|
+
|
|
254
455
|
def slab_pointer
|
|
255
456
|
@slab_pointer ||= @handle[SLAB_NAME]
|
|
256
457
|
end
|
|
@@ -280,6 +481,35 @@ class CArray
|
|
|
280
481
|
}).pack("d*")
|
|
281
482
|
end
|
|
282
483
|
|
|
484
|
+
# The `integers` slots the kernel reads: the signed captures, then the
|
|
485
|
+
# unsigned ones, then the extents an index is checked against. A uint64
|
|
486
|
+
# is packed as the bits it is -- `pack("q")` would take 2**63 modulo the
|
|
487
|
+
# width and say nothing -- and the kernel casts the slot back, so the
|
|
488
|
+
# two directives are what makes one buffer carry both widths.
|
|
489
|
+
#
|
|
490
|
+
# One method because both loops pack it, and the same buffer packed in
|
|
491
|
+
# two places is what `packed_reals` is one method about.
|
|
492
|
+
def packed_integers (scalar_values, array_values)
|
|
493
|
+
packed_scalar_integers(scalar_values) + packed_extents(array_values)
|
|
494
|
+
end
|
|
495
|
+
|
|
496
|
+
# The captured half, which is different every time a caller steps a
|
|
497
|
+
# counter, and the arrays' half, which is the shape and travels in the
|
|
498
|
+
# plan. Two methods because `#call` packs the first and takes the
|
|
499
|
+
# second from what it prepared earlier; `#sweep` still wants both.
|
|
500
|
+
def packed_scalar_integers (scalar_values)
|
|
501
|
+
@integers.map { |name| Integer(scalar_values.fetch(name)) }.pack("q*") +
|
|
502
|
+
@unsigned_integers.map { |name|
|
|
503
|
+
Integer(scalar_values.fetch(name))
|
|
504
|
+
}.pack("Q*")
|
|
505
|
+
end
|
|
506
|
+
|
|
507
|
+
def packed_extents (array_values)
|
|
508
|
+
@extent_slots.map { |name, axis|
|
|
509
|
+
array_values.fetch(name).dim[axis]
|
|
510
|
+
}.pack("q*")
|
|
511
|
+
end
|
|
512
|
+
|
|
283
513
|
def address_buffer (name, array)
|
|
284
514
|
@address_parameters.fetch(name).each do |parameter|
|
|
285
515
|
wanted = CDeclaration::DATA_TYPES.fetch(parameter.element.fiddle)
|
|
@@ -318,7 +548,7 @@ class CArray
|
|
|
318
548
|
packed.empty? ? "\0" * 8 : packed
|
|
319
549
|
end
|
|
320
550
|
|
|
321
|
-
# The
|
|
551
|
+
# The xfer tier transfers a box rather than the whole view. The box
|
|
322
552
|
# is the loop range grown by how far the kernel reaches from the cell it
|
|
323
553
|
# is on -- per array, because two arrays in one kernel need not be read
|
|
324
554
|
# at the same offsets, and per axis, because they need not be read at
|
|
@@ -376,13 +606,117 @@ class CArray
|
|
|
376
606
|
ranges[name] = covered_span(bounds[axis])
|
|
377
607
|
end
|
|
378
608
|
flat = bounds.flatten
|
|
379
|
-
|
|
380
|
-
|
|
381
|
-
|
|
609
|
+
# In source order, outermost first, so an index a range is written
|
|
610
|
+
# over has its own range settled before it is read.
|
|
611
|
+
@inner_ranges.each do |name, (from, to, step)|
|
|
612
|
+
step ||= 1
|
|
613
|
+
first = span_of(from, scalar_values, flat, ranges, name)
|
|
614
|
+
last = span_of(to, scalar_values, flat, ranges, name)
|
|
615
|
+
# The widest pass the range could describe: counting up, that is the
|
|
616
|
+
# earliest start against the latest end, and counting down it is the
|
|
617
|
+
# other way round.
|
|
618
|
+
start, limit = step.positive? ? [first.first, last.last]
|
|
619
|
+
: [first.last, last.first]
|
|
620
|
+
# A step that skips cells lands on the ones a pass started on, and
|
|
621
|
+
# where the start moves so do they: `(p...8).step(3)` visits 0, 3, 6
|
|
622
|
+
# from one start and 1, 4, 7 from the next. Stepped out from the
|
|
623
|
+
# earliest start alone, the reach stopped at 6 and a pass wrote at
|
|
624
|
+
# 7. So with a start that moves, the span is every cell between
|
|
625
|
+
# the widest start and the widest end, which every pass lies within.
|
|
626
|
+
step = step <=> 0 if first.first != first.last
|
|
627
|
+
ranges[name] = covered_span([start, limit, step])
|
|
382
628
|
end
|
|
383
629
|
ranges
|
|
384
630
|
end
|
|
385
631
|
|
|
632
|
+
# A range written over another index is read at its widest (see
|
|
633
|
+
# #span_of), so the number in the message above can be one no pass
|
|
634
|
+
# actually starts or ends at. Saying which index it was written over is
|
|
635
|
+
# what tells a reader whether to narrow the range or the array's use --
|
|
636
|
+
# without it the message names a bound the loop never reaches.
|
|
637
|
+
def widest_reading (index, spelled)
|
|
638
|
+
over = range_indices(index)
|
|
639
|
+
return "" if over.empty?
|
|
640
|
+
". The range on `#{spelled}` is written over " \
|
|
641
|
+
"#{over.map { |name| "`#{name}`" }.join(' and ')}, and a range over " \
|
|
642
|
+
"another index is read here at its widest -- as far as any value of " \
|
|
643
|
+
"#{over.size == 1 ? 'it' : 'them'} could take it, which may be " \
|
|
644
|
+
"further than it goes on any one pass"
|
|
645
|
+
end
|
|
646
|
+
|
|
647
|
+
# The indices an inner loop's range mentions, by the name the block
|
|
648
|
+
# wrote for each.
|
|
649
|
+
def range_indices (index)
|
|
650
|
+
range = @inner_ranges[index]
|
|
651
|
+
return [] unless range
|
|
652
|
+
from, to, = range
|
|
653
|
+
[from, to].flat_map { |node| mentioned_indices(node) }.uniq
|
|
654
|
+
.map { |name| @index_sources.fetch(name, name) }
|
|
655
|
+
end
|
|
656
|
+
|
|
657
|
+
def mentioned_indices (node)
|
|
658
|
+
return [] unless node.is_a?(Node)
|
|
659
|
+
return [node.name] if node.is_a?(IndexVariable)
|
|
660
|
+
node.children.flat_map { |child| mentioned_indices(child) }
|
|
661
|
+
end
|
|
662
|
+
|
|
663
|
+
# What an index range covers, as [lowest, highest] inclusive.
|
|
664
|
+
#
|
|
665
|
+
# A range written over another index -- `(p+1...3)`, the shape a
|
|
666
|
+
# triangular loop takes -- has no single pair of numbers to be: `k`
|
|
667
|
+
# starts somewhere else for every `p`. What the two callers of this
|
|
668
|
+
# want is where the index can reach, not what it is on a given pass:
|
|
669
|
+
# `verify_bounds` asks whether a subscript can leave its array, and
|
|
670
|
+
# `region_box` asks what a stencil touches. So the answer is an
|
|
671
|
+
# interval, and a range over another index is read at its widest.
|
|
672
|
+
#
|
|
673
|
+
# Widest is the safe direction. It covers every pass the loop can take
|
|
674
|
+
# and then some, so a subscript this says is inside really is; what it
|
|
675
|
+
# costs is that a reach nothing actually makes can still be refused.
|
|
676
|
+
#
|
|
677
|
+
# A range that mentions no index gives an interval of one number, which
|
|
678
|
+
# is what `evaluate` gave before, so those kernels are unchanged.
|
|
679
|
+
def span_of (node, scalars, flat_bounds, ranges, index)
|
|
680
|
+
case node
|
|
681
|
+
when IntegerLiteral then [node.value, node.value]
|
|
682
|
+
when CaptureRead then value = Integer(scalars.fetch(node.name))
|
|
683
|
+
[value, value]
|
|
684
|
+
when BoundsValue then value = flat_bounds.fetch(node.slot)
|
|
685
|
+
[value, value]
|
|
686
|
+
when IndexVariable
|
|
687
|
+
# Held half-open, as the loops are; the last index it visits is one
|
|
688
|
+
# below the end.
|
|
689
|
+
low, high = ranges.fetch(node.name) do
|
|
690
|
+
raise Unsupported,
|
|
691
|
+
"the range on `#{index}` is written over `#{node.name}`, " \
|
|
692
|
+
"which is not an index around it"
|
|
693
|
+
end
|
|
694
|
+
[low, high - 1]
|
|
695
|
+
when UnaryMinus
|
|
696
|
+
low, high = span_of(node.operand, scalars, flat_bounds, ranges, index)
|
|
697
|
+
[-high, -low]
|
|
698
|
+
when BinaryOperation
|
|
699
|
+
left = span_of(node.left, scalars, flat_bounds, ranges, index)
|
|
700
|
+
right = span_of(node.right, scalars, flat_bounds, ranges, index)
|
|
701
|
+
case node.operator
|
|
702
|
+
when :+ then [left.first + right.first, left.last + right.last]
|
|
703
|
+
when :- then [left.first - right.last, left.last - right.first]
|
|
704
|
+
when :*
|
|
705
|
+
# Four corners, because either interval may straddle zero.
|
|
706
|
+
corners = [left.first * right.first, left.first * right.last,
|
|
707
|
+
left.last * right.first, left.last * right.last]
|
|
708
|
+
[corners.min, corners.max]
|
|
709
|
+
else
|
|
710
|
+
raise Unsupported,
|
|
711
|
+
"an inner loop's range is built from `+`, `-` and `*` only"
|
|
712
|
+
end
|
|
713
|
+
else
|
|
714
|
+
raise Unsupported,
|
|
715
|
+
"an inner loop's range is an integer expression over " \
|
|
716
|
+
"literals, captured scalars and the indices around it"
|
|
717
|
+
end
|
|
718
|
+
end
|
|
719
|
+
|
|
386
720
|
# The half-open span an axis actually touches. A step may skip cells,
|
|
387
721
|
# but every cell it lands on has to exist, so the span runs from the
|
|
388
722
|
# first index to the last one visited.
|
|
@@ -463,19 +797,22 @@ class CArray
|
|
|
463
797
|
walkers.each do |index, offsets|
|
|
464
798
|
low, high = ranges.fetch(index)
|
|
465
799
|
next if low >= high
|
|
800
|
+
spelled = @index_sources.fetch(index, index)
|
|
466
801
|
minimum, maximum = offset_span(offsets, scalars)
|
|
467
802
|
if low + minimum < 0
|
|
468
803
|
raise Unsupported,
|
|
469
804
|
"`#{name}` is indexed at " \
|
|
470
|
-
"#{offset_text(name,
|
|
471
|
-
"the range on `#{
|
|
805
|
+
"#{offset_text(name, spelled, minimum)}, so " \
|
|
806
|
+
"the range on `#{spelled}` cannot start at #{low}" +
|
|
807
|
+
widest_reading(index, spelled)
|
|
472
808
|
end
|
|
473
809
|
if high - 1 + maximum > extent - 1
|
|
474
810
|
raise Unsupported,
|
|
475
811
|
"`#{name}` is indexed at " \
|
|
476
|
-
"#{offset_text(name,
|
|
477
|
-
"the range on `#{
|
|
478
|
-
"for an extent of #{extent}"
|
|
812
|
+
"#{offset_text(name, spelled, maximum)}, so " \
|
|
813
|
+
"the range on `#{spelled}` cannot end at #{high} " \
|
|
814
|
+
"for an extent of #{extent}" +
|
|
815
|
+
widest_reading(index, spelled)
|
|
479
816
|
end
|
|
480
817
|
end
|
|
481
818
|
end
|