carray-jit 0.1.2 → 0.1.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +771 -3
- data/README.md +7 -6
- data/carray-jit.gemspec +1 -3
- data/docs/00_Introduction.md +4 -3
- data/docs/01_GettingStarted.md +1 -1
- data/docs/02_KernelShapes.md +93 -14
- data/docs/03_SupportedFeatures.md +582 -26
- data/docs/04_Compiling.md +33 -6
- data/docs/05_DesignNotes.md +3 -3
- data/docs/06_Cheatsheet.md +198 -5
- data/docs/07_StepByStep.ja.md +534 -0
- data/docs/07_StepByStep.md +535 -0
- data/examples/README.md +12 -0
- data/examples/applications/alarm.rb +121 -0
- data/examples/applications/collatz.rb +105 -0
- data/examples/applications/cubic_spline.rb +331 -0
- data/examples/applications/dithering.rb +144 -0
- data/examples/applications/group_stats.rb +115 -0
- data/examples/applications/lookup.rb +126 -0
- data/examples/applications/median_filter.rb +153 -0
- data/examples/applications/parcel_ascent.rb +220 -0
- data/examples/applications/point_in_polygon.rb +111 -0
- data/examples/applications/random_walk.rb +98 -0
- data/examples/applications/van_der_pol.rb +186 -0
- data/examples/applications/wet_bulb.rb +140 -0
- data/examples/features/10_complex.rb +14 -4
- data/examples/features/15_loops.rb +7 -1
- data/lib/carray/jit/access.rb +14 -0
- data/lib/carray/jit/analyzer.rb +2077 -136
- data/lib/carray/jit/block_reader.rb +37 -6
- data/lib/carray/jit/c_function.rb +613 -76
- data/lib/carray/jit/c_generator.rb +1595 -156
- data/lib/carray/jit/call.rb +68 -0
- data/lib/carray/jit/compiler.rb +75 -11
- data/lib/carray/jit/kernel.rb +369 -32
- data/lib/carray/jit/node.rb +359 -9
- data/lib/carray/jit/sorting_networks.rb +182 -0
- data/lib/carray/jit/type_assignment.rb +371 -34
- data/lib/carray/jit/version.rb +1 -1
- data/lib/carray/jit.rb +560 -64
- metadata +22 -8
- data/ext/carray_jit_access/carray_jit_access.c +0 -460
- data/ext/carray_jit_access/extconf.rb +0 -8
|
@@ -126,6 +126,15 @@ class CArray
|
|
|
126
126
|
# would. Only the ones a kernel actually uses are emitted, and the
|
|
127
127
|
# reason each exists is at emit_complex_binary.
|
|
128
128
|
COMPLEX_HELPERS = {
|
|
129
|
+
"complex_finite" => <<~C,
|
|
130
|
+
/* Complex#finite? is both parts finite, which is what Ruby asks
|
|
131
|
+
and is not what isfinite of a complex would mean. */
|
|
132
|
+
static inline int
|
|
133
|
+
carray_jit_complex_finite (double _Complex z)
|
|
134
|
+
{
|
|
135
|
+
return isfinite(creal(z)) && isfinite(cimag(z));
|
|
136
|
+
}
|
|
137
|
+
C
|
|
129
138
|
"complex_add_real" => <<~C,
|
|
130
139
|
/* Ruby leaves the imaginary part of `z + x` exactly as it was,
|
|
131
140
|
rather than adding the real operand's zero to it. */
|
|
@@ -159,9 +168,61 @@ class CArray
|
|
|
159
168
|
return CMPLX(creal(z), y + cimag(z));
|
|
160
169
|
}
|
|
161
170
|
C
|
|
171
|
+
"safe_multiply" => <<~C,
|
|
172
|
+
/* complex.c's safe_mul, which every part of a Ruby complex product
|
|
173
|
+
goes through: a zero against a number that is not a zero or a
|
|
174
|
+
NaN meets only that number's sign, so a zero times an infinity is
|
|
175
|
+
a signed zero rather than C's NaN. C's own complex `*` recovers
|
|
176
|
+
from infinities by Annex G's rules instead, which are not these. */
|
|
177
|
+
static inline double
|
|
178
|
+
carray_jit_safe_multiply (double a, double b)
|
|
179
|
+
{
|
|
180
|
+
if ( a != 0 && b == 0 && ! isnan(a) ) a = signbit(a) ? -1.0 : 1.0;
|
|
181
|
+
else if ( b != 0 && a == 0 && ! isnan(b) ) b = signbit(b) ? -1.0 : 1.0;
|
|
182
|
+
return a * b;
|
|
183
|
+
}
|
|
184
|
+
C
|
|
185
|
+
"complex_multiply" => <<~C,
|
|
186
|
+
/* Ruby's `a * b` for two Complex numbers, part by part in
|
|
187
|
+
complex.c's order: comp_mul. safe_mul differs from a plain
|
|
188
|
+
product only for a zero against an infinity, which a plain
|
|
189
|
+
product answers NaN -- so the plain parts are the answer unless
|
|
190
|
+
one of them is a NaN, and only then is it worked out again the
|
|
191
|
+
safe way. C's own `*` takes the same shape, and costs the same. */
|
|
192
|
+
static inline double _Complex
|
|
193
|
+
carray_jit_complex_multiply (double _Complex a, double _Complex b)
|
|
194
|
+
{
|
|
195
|
+
double are = creal(a), aim = cimag(a);
|
|
196
|
+
double bre = creal(b), bim = cimag(b);
|
|
197
|
+
double re = are * bre - aim * bim, im = are * bim + aim * bre;
|
|
198
|
+
if ( isnan(re) || isnan(im) ) {
|
|
199
|
+
re = carray_jit_safe_multiply(are, bre) - carray_jit_safe_multiply(aim, bim);
|
|
200
|
+
im = carray_jit_safe_multiply(are, bim) + carray_jit_safe_multiply(aim, bre);
|
|
201
|
+
}
|
|
202
|
+
return CMPLX(re, im);
|
|
203
|
+
}
|
|
204
|
+
C
|
|
205
|
+
"real_multiply_complex" => <<~C,
|
|
206
|
+
/* `x * z` coerces x to a Complex whose imaginary part is an exact
|
|
207
|
+
zero, and multiplies in full. That zero meets safe_mul as any
|
|
208
|
+
zero does, so it is written as 0.0 here and gives the same parts;
|
|
209
|
+
only an addition treats an exact zero differently, and here the
|
|
210
|
+
zeros are multiplied first. */
|
|
211
|
+
static inline double _Complex
|
|
212
|
+
carray_jit_real_multiply_complex (double x, double _Complex b)
|
|
213
|
+
{
|
|
214
|
+
double bre = creal(b), bim = cimag(b);
|
|
215
|
+
double re = x * bre - 0.0 * bim, im = x * bim + 0.0 * bre;
|
|
216
|
+
if ( isnan(re) || isnan(im) ) {
|
|
217
|
+
re = carray_jit_safe_multiply(x, bre) - carray_jit_safe_multiply(0.0, bim);
|
|
218
|
+
im = carray_jit_safe_multiply(x, bim) + carray_jit_safe_multiply(0.0, bre);
|
|
219
|
+
}
|
|
220
|
+
return CMPLX(re, im);
|
|
221
|
+
}
|
|
222
|
+
C
|
|
162
223
|
"complex_mul_real" => <<~C,
|
|
163
|
-
/* `z * x` scales each part
|
|
164
|
-
|
|
224
|
+
/* `z * x` scales each part, with no safe_mul: Ruby's Complex#*
|
|
225
|
+
takes a real operand apart and multiplies each part by it. */
|
|
165
226
|
static inline double _Complex
|
|
166
227
|
carray_jit_complex_mul_real (double _Complex z, double x)
|
|
167
228
|
{
|
|
@@ -256,8 +317,10 @@ class CArray
|
|
|
256
317
|
BORDER_RULES = [:zero, :clamp, :wrap].freeze
|
|
257
318
|
|
|
258
319
|
def initialize (analyzer, storage_types, scalar_types, c_functions: {},
|
|
320
|
+
randoms: {},
|
|
259
321
|
masked: false, reassociate: false,
|
|
260
|
-
steps: nil, origin: nil, block_source: nil, border: nil
|
|
322
|
+
steps: nil, origin: nil, block_source: nil, border: nil,
|
|
323
|
+
scalar_parameters: [])
|
|
261
324
|
@masked = masked
|
|
262
325
|
# What a window read gets where it falls off the array. Nil for every
|
|
263
326
|
# kernel but a stencil's, and for a stencil whose border is the frame
|
|
@@ -282,25 +345,47 @@ class CArray
|
|
|
282
345
|
position += analyzer.array_ranks.fetch(array, analyzer.rank)
|
|
283
346
|
end
|
|
284
347
|
@array_ranks = analyzer.array_ranks
|
|
285
|
-
|
|
286
|
-
|
|
348
|
+
# A compiled function's scalars are the parameters of its own C
|
|
349
|
+
# signature: they arrive by the ABI, in whatever type the declaration
|
|
350
|
+
# named, so no buffer carries them and they take no bus. Set aside
|
|
351
|
+
# here, which leaves the check below about captures -- what it was
|
|
352
|
+
# always about.
|
|
353
|
+
carried_names = analyzer.scalar_names - scalar_parameters
|
|
354
|
+
@reals = carried_names.select { |name| scalar_types[name] == :double }.sort
|
|
355
|
+
@integers = carried_names.select { |name| scalar_types[name] == :int64 }.sort
|
|
287
356
|
# A captured Complex rides in the reals buffer as its two parts, so
|
|
288
357
|
# that the kernel signature stays the one shape every kernel has.
|
|
289
|
-
@complexes =
|
|
290
|
-
#
|
|
291
|
-
#
|
|
292
|
-
#
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
|
|
358
|
+
@complexes = carried_names.select { |name| scalar_types[name] == :complex }.sort
|
|
359
|
+
# And a captured uint64 rides in the integers buffer, which is the
|
|
360
|
+
# same eight bytes read the other way: the caller packs the value's
|
|
361
|
+
# bits and the kernel casts the slot back, so nothing is lost and the
|
|
362
|
+
# signature keeps its three buffers. Behind the signed ones, in the
|
|
363
|
+
# order the caller packs them.
|
|
364
|
+
@unsigned_integers =
|
|
365
|
+
carried_names.select { |name| scalar_types[name] == :uint64 }.sort
|
|
366
|
+
# Three buffers and no fourth: a capture whose type rides in none of
|
|
367
|
+
# them would be packed into nothing and read as whatever the slot
|
|
368
|
+
# held, so it is caught here rather than at the cell it computes
|
|
369
|
+
# wrongly.
|
|
370
|
+
carried = @reals.size + @integers.size + @complexes.size +
|
|
371
|
+
@unsigned_integers.size
|
|
372
|
+
unless carried == carried_names.size
|
|
373
|
+
missing = carried_names - @reals - @integers - @complexes -
|
|
374
|
+
@unsigned_integers
|
|
296
375
|
raise Error,
|
|
297
376
|
"captured #{missing.join(", ")} travel in none of the kernel's " \
|
|
298
|
-
"
|
|
377
|
+
"scalar buffers; a computation type was added without a " \
|
|
299
378
|
"way to hand a value of it to the C"
|
|
300
379
|
end
|
|
301
380
|
# Only the ones the block actually called, in a fixed order, because
|
|
302
381
|
# the caller packs the addresses into `functions` by this order.
|
|
303
382
|
@c_functions = analyzer.c_function_names.sort.to_h { |name| [name, c_functions.fetch(name)] }
|
|
383
|
+
# The generators the block drew from, by the name it drew through.
|
|
384
|
+
# Unlike a C function, one takes no slot in any buffer: its C is
|
|
385
|
+
# pasted and its state is an address array, so what is kept here is
|
|
386
|
+
# only which generator each name is -- the kind, to paste, and the
|
|
387
|
+
# symbol to call.
|
|
388
|
+
@randoms = analyzer.random_names.to_h { |name| [name, randoms.fetch(name)] }
|
|
304
389
|
# A function written here is pasted into this kernel and called by its
|
|
305
390
|
# symbol; a borrowed one arrives as an address. Only the second kind
|
|
306
391
|
# takes a slot in `functions`, so it is that hash the caller packs by.
|
|
@@ -330,10 +415,14 @@ class CArray
|
|
|
330
415
|
@uses_clamp = false
|
|
331
416
|
@uses_wrap = false
|
|
332
417
|
@uses_real_arg = false
|
|
418
|
+
@clamp_types = []
|
|
419
|
+
@gamma_types = []
|
|
333
420
|
@complex_helpers = []
|
|
334
421
|
@uses_floor_divide = false
|
|
335
422
|
@uses_floor_modulo = false
|
|
336
423
|
@uses_integer_power = false
|
|
424
|
+
@uses_integer_abs = false
|
|
425
|
+
@uses_whole_number = false
|
|
337
426
|
@uses_unsigned_divide = false
|
|
338
427
|
@uses_unsigned_modulo = false
|
|
339
428
|
@uses_unsigned_power = false
|
|
@@ -350,17 +439,28 @@ class CArray
|
|
|
350
439
|
# this kernel is what answers for them. The codes agree because they
|
|
351
440
|
# are taken from the messages, not counted off.
|
|
352
441
|
@raise_messages = {}
|
|
353
|
-
|
|
442
|
+
pasted_closure.each do |function|
|
|
354
443
|
(function.raise_messages || {}).each do |code, message|
|
|
355
444
|
register_raise(code, message)
|
|
356
445
|
end
|
|
357
446
|
end
|
|
447
|
+
# The intrinsic helpers the body asked for, as [name, storage, cells]
|
|
448
|
+
# -- a list used as a set, the way @clamp_types is. A pasted body's
|
|
449
|
+
# are merged into it by #preamble.
|
|
450
|
+
@intrinsic_needs = []
|
|
451
|
+
# Whether some body pasted into this file clears an array of its own.
|
|
452
|
+
@pasted_clears_a_local_array = false
|
|
358
453
|
@position_temporaries = {}.compare_by_identity
|
|
454
|
+
# The positions a write is being guarded by, by axis, while its body
|
|
455
|
+
# is emitted.
|
|
456
|
+
@guarded_axes = {}.compare_by_identity
|
|
359
457
|
@origin = origin
|
|
360
458
|
@block_source = block_source
|
|
459
|
+
refuse_generated_indices
|
|
361
460
|
end
|
|
362
461
|
|
|
363
|
-
attr_reader :arrays, :reals, :integers, :complexes, :
|
|
462
|
+
attr_reader :arrays, :reals, :integers, :complexes, :unsigned_integers,
|
|
463
|
+
:masked, :extent_slots,
|
|
364
464
|
:c_functions, :pasted_functions, :address_functions,
|
|
365
465
|
:address_arrays, :address_parameters,
|
|
366
466
|
# The messages `raise` in the block gave, by the code a cell
|
|
@@ -373,6 +473,14 @@ class CArray
|
|
|
373
473
|
@uses_error_flag
|
|
374
474
|
end
|
|
375
475
|
|
|
476
|
+
# Whether some array the body made is one of the zeroed spellings, which
|
|
477
|
+
# is what `<string.h>` is included for. A pasted function's body is
|
|
478
|
+
# emitted into this file too, and when a function body may make one of
|
|
479
|
+
# its own this has to ask the pasted ones as well.
|
|
480
|
+
def clears_a_local_array?
|
|
481
|
+
@analyzer.clears_a_local_array? || @pasted_clears_a_local_array
|
|
482
|
+
end
|
|
483
|
+
|
|
376
484
|
def generate
|
|
377
485
|
strided = build_body(false)
|
|
378
486
|
contiguous = build_body(true)
|
|
@@ -412,8 +520,8 @@ class CArray
|
|
|
412
520
|
@own_symbol = name
|
|
413
521
|
@in_function = true
|
|
414
522
|
@error_parameter = error_parameter
|
|
523
|
+
@body_reports = false
|
|
415
524
|
@contiguous = true
|
|
416
|
-
@declared_locals = {}
|
|
417
525
|
@temporary_count = 0
|
|
418
526
|
@carried_masks = []
|
|
419
527
|
statements = @analyzer.body.statements
|
|
@@ -422,6 +530,7 @@ class CArray
|
|
|
422
530
|
@returns_nothing = return_type.nil?
|
|
423
531
|
held = @returns_nothing ? statements : statements[0..-2]
|
|
424
532
|
lines = held.map { |statement| emit_statement(statement, " ") }.join
|
|
533
|
+
lines = local_declarations(@analyzer.body, " ") + lines
|
|
425
534
|
value = @returns_nothing ? nil : emit(statements.last, return_type)
|
|
426
535
|
parameters = parameter_names.zip(parameter_c_types)
|
|
427
536
|
.map { |parameter, type| type.declare(parameter) }
|
|
@@ -434,26 +543,58 @@ class CArray
|
|
|
434
543
|
# Kept apart from the file it is compiled in: the definition alone is
|
|
435
544
|
# what a kernel pastes into its own translation unit, where the
|
|
436
545
|
# includes are already written and the helpers are shared.
|
|
437
|
-
|
|
438
|
-
|
|
439
|
-
|
|
440
|
-
preamble + flag +
|
|
546
|
+
opening = "#{return_c_type}\n#{name} (#{parameters.join(', ')})\n{\n"
|
|
547
|
+
body = lines + (@returns_nothing ? "}\n" : " return #{value};\n}\n")
|
|
548
|
+
@function_definition = opening + body
|
|
549
|
+
preamble + flag + opening + standing_gate + body
|
|
441
550
|
end
|
|
442
551
|
|
|
443
552
|
attr_reader :function_definition
|
|
444
553
|
|
|
554
|
+
# A body standing on its own does no more work once its flag stands.
|
|
555
|
+
#
|
|
556
|
+
# The `!ERROR_FLAG` at each `raise` keeps a later report from overwriting
|
|
557
|
+
# an earlier one, and that is enough for a caller that stops: a kernel
|
|
558
|
+
# ends its loop, and `CFunction#call` was one call to begin with. A
|
|
559
|
+
# library handed the address does not stop. It calls again, and without
|
|
560
|
+
# this the body skips the report it has already made and answers as
|
|
561
|
+
# though nothing had happened -- so an adaptive routine converges on
|
|
562
|
+
# values that are fiction, which is worse than a wrong answer because it
|
|
563
|
+
# looks like a right one.
|
|
564
|
+
#
|
|
565
|
+
# What the gate says is that the body does no more work. It does not
|
|
566
|
+
# say what the call is worth: the flag is the contract, and the zero is
|
|
567
|
+
# a courtesy to a caller who never looks at it. A `void` body has no
|
|
568
|
+
# courtesy to offer and leaves its out-parameters alone, which is the
|
|
569
|
+
# same refusal to work.
|
|
570
|
+
#
|
|
571
|
+
# It goes in the file and not in the definition, so that the form a
|
|
572
|
+
# kernel pastes cannot pick it up whichever way the two forms are
|
|
573
|
+
# generated. A kernel reads its slot once, after a loop it stopped
|
|
574
|
+
# itself, and has no use for it.
|
|
575
|
+
def standing_gate
|
|
576
|
+
return "" unless @uses_error_flag && !@error_parameter
|
|
577
|
+
" if ( #{ERROR_FLAG} ) #{@returns_nothing ? "return;" : "return 0;"}\n\n"
|
|
578
|
+
end
|
|
579
|
+
|
|
445
580
|
# What the body took from the preamble, so that a definition pasted
|
|
446
581
|
# somewhere else can be given the same helpers. Only the ones a pasted
|
|
447
582
|
# function can want are here: the rest -- the index check, the two
|
|
448
583
|
# flooring helpers -- report through the error flag, and a function that
|
|
449
584
|
# touches the flag is not pasted at all.
|
|
450
585
|
def helper_needs
|
|
451
|
-
{ :
|
|
586
|
+
{ :intrinsics => @intrinsic_needs.dup,
|
|
587
|
+
:clears_a_local_array => clears_a_local_array?,
|
|
588
|
+
:integer_power => @uses_integer_power,
|
|
589
|
+
:integer_abs => @uses_integer_abs,
|
|
590
|
+
:whole_number => @uses_whole_number,
|
|
452
591
|
:unsigned_divide => @uses_unsigned_divide,
|
|
453
592
|
:unsigned_modulo => @uses_unsigned_modulo,
|
|
454
593
|
:unsigned_power => @uses_unsigned_power,
|
|
455
594
|
:floor_modulo_float => @uses_floor_modulo_float,
|
|
456
595
|
:real_arg => @uses_real_arg,
|
|
596
|
+
:clamp => @clamp_types.dup,
|
|
597
|
+
:gamma => @gamma_types.dup,
|
|
457
598
|
:complex => @complex_helpers.dup,
|
|
458
599
|
:index_check => @uses_index_check,
|
|
459
600
|
:floor_divide => @uses_floor_divide,
|
|
@@ -547,27 +688,64 @@ class CArray
|
|
|
547
688
|
# A pasted body wants the same helpers here that it had in its own
|
|
548
689
|
# file, and it is emitted below them, so this is asked before any of
|
|
549
690
|
# them is written out.
|
|
550
|
-
|
|
691
|
+
pasted_closure.each do |function|
|
|
551
692
|
needs = function.helpers || {}
|
|
552
693
|
@uses_integer_power ||= needs[:integer_power]
|
|
694
|
+
@uses_integer_abs ||= needs[:integer_abs]
|
|
695
|
+
@uses_whole_number ||= needs[:whole_number]
|
|
553
696
|
@uses_unsigned_divide ||= needs[:unsigned_divide]
|
|
554
697
|
@uses_unsigned_modulo ||= needs[:unsigned_modulo]
|
|
555
698
|
@uses_unsigned_power ||= needs[:unsigned_power]
|
|
556
699
|
@uses_floor_modulo_float ||= needs[:floor_modulo_float]
|
|
557
700
|
@uses_real_arg ||= needs[:real_arg]
|
|
701
|
+
@clamp_types |= needs[:clamp] || []
|
|
702
|
+
@gamma_types |= needs[:gamma] || []
|
|
558
703
|
@complex_helpers |= needs[:complex] || []
|
|
559
704
|
@uses_index_check ||= needs[:index_check]
|
|
560
705
|
@uses_floor_divide ||= needs[:floor_divide]
|
|
561
706
|
@uses_floor_modulo ||= needs[:floor_modulo]
|
|
562
|
-
|
|
563
|
-
|
|
564
|
-
|
|
707
|
+
# A union rather than an or: two bodies may want a `sort` of one
|
|
708
|
+
# type and length and a `sum` of another, and the preamble owes one
|
|
709
|
+
# helper for each. The same shape `:clamp` and `:gamma` take.
|
|
710
|
+
@intrinsic_needs |= needs[:intrinsics] || []
|
|
711
|
+
# And whether anything anywhere in this file clears an array of its
|
|
712
|
+
# own, which is what `<string.h>` is included for. A pasted body's
|
|
713
|
+
# `memset` is in this translation unit, so it is this file's header.
|
|
714
|
+
@pasted_clears_a_local_array ||= !!needs[:clears_a_local_array]
|
|
715
|
+
end
|
|
716
|
+
# `stddef.h` is here for the signature's sake rather than the body's:
|
|
717
|
+
# a declaration may be written with `size_t` or `ptrdiff_t`, and that
|
|
718
|
+
# is the header those come from. `stdio.h` happens to declare
|
|
719
|
+
# `size_t` as well, which is why the omission went unnoticed until
|
|
720
|
+
# `ptrdiff_t` was written down.
|
|
721
|
+
text = +"#include <stdint.h>\n#include <stddef.h>\n" \
|
|
722
|
+
"#include <math.h>\n#include <complex.h>\n" \
|
|
723
|
+
"#include <stdio.h>\n"
|
|
724
|
+
# `memset`, and only where a body clears an array of its own. A
|
|
725
|
+
# kernel that clears nothing generates what it always did.
|
|
726
|
+
text << "#include <string.h>\n" if clears_a_local_array?
|
|
727
|
+
# malloc and free, for the arrays this kernel allocates. libc, the
|
|
728
|
+
# same ground `memset` above stands on: it needs no GVL and touches
|
|
729
|
+
# no Ruby value, which is what kept `ALLOCV` out of here.
|
|
730
|
+
text << "#include <stdlib.h>\n" unless heap_arrays.empty?
|
|
731
|
+
text << "\n"
|
|
565
732
|
unless @address_functions.empty?
|
|
566
|
-
|
|
567
|
-
|
|
568
|
-
|
|
569
|
-
|
|
570
|
-
text <<
|
|
733
|
+
if @in_function
|
|
734
|
+
# A body standing on its own calls a borrowed function by its
|
|
735
|
+
# name, so what it needs is that function's own declaration --
|
|
736
|
+
# the one its author wrote -- and a linker.
|
|
737
|
+
text << "/* Declared where they are called from, and resolved " \
|
|
738
|
+
"by the linker\n or the loader. */\n"
|
|
739
|
+
@address_functions.each_value do |function|
|
|
740
|
+
text << "#{function.prototype};\n"
|
|
741
|
+
end
|
|
742
|
+
else
|
|
743
|
+
text << "/* The C functions the block called. They arrive as\n" \
|
|
744
|
+
" addresses rather than by linkage, so nothing here says\n" \
|
|
745
|
+
" which library they came from. */\n"
|
|
746
|
+
@address_functions.each_key do |name|
|
|
747
|
+
text << @address_functions.fetch(name).c_declaration(c_function_type_name(name)) << "\n"
|
|
748
|
+
end
|
|
571
749
|
end
|
|
572
750
|
text << "\n"
|
|
573
751
|
end
|
|
@@ -589,8 +767,12 @@ class CArray
|
|
|
589
767
|
computing wide and narrowing afterwards stops agreeing with
|
|
590
768
|
computing narrow. */
|
|
591
769
|
static inline float
|
|
592
|
-
carray_jit_floor_modulo_float (float numerator, float denominator)
|
|
770
|
+
carray_jit_floor_modulo_float (float numerator, float denominator, int32_t *error)
|
|
593
771
|
{
|
|
772
|
+
if ( denominator == 0 ) {
|
|
773
|
+
if ( error ) *error = 1;
|
|
774
|
+
return 0;
|
|
775
|
+
}
|
|
594
776
|
float remainder = fmodf(numerator, denominator);
|
|
595
777
|
if ( remainder != 0 ) {
|
|
596
778
|
if ( (remainder < 0) != (denominator < 0) ) remainder += denominator;
|
|
@@ -677,6 +859,54 @@ class CArray
|
|
|
677
859
|
|
|
678
860
|
C
|
|
679
861
|
end
|
|
862
|
+
if @uses_whole_number
|
|
863
|
+
text << <<~C
|
|
864
|
+
/* `floor`, `ceil`, `round` and `truncate` on a Float answer an
|
|
865
|
+
Integer in Ruby, and a NaN or an infinity has none to answer:
|
|
866
|
+
Ruby raises FloatDomainError. A number past int64 has one, but
|
|
867
|
+
not in an int64_t, where the cast is undefined -- CArray raises
|
|
868
|
+
RangeError storing such an Integer, and so does this. The real
|
|
869
|
+
one is for a value that stays a number: 1e20.floor into a
|
|
870
|
+
float64 cell is 1e20 in Ruby, and does not pass through int64. */
|
|
871
|
+
static inline double
|
|
872
|
+
carray_jit_whole_real (double value, int32_t *error)
|
|
873
|
+
{
|
|
874
|
+
if ( isnan(value) ) {
|
|
875
|
+
if ( error ) *error = #{FLOAT_NAN_CODE};
|
|
876
|
+
}
|
|
877
|
+
else if ( isinf(value) ) {
|
|
878
|
+
if ( error ) *error = value > 0 ? #{FLOAT_INFINITY_CODE} : #{FLOAT_NEGATIVE_INFINITY_CODE};
|
|
879
|
+
}
|
|
880
|
+
return value;
|
|
881
|
+
}
|
|
882
|
+
|
|
883
|
+
static inline int64_t
|
|
884
|
+
carray_jit_whole_int64 (double value, int32_t *error)
|
|
885
|
+
{
|
|
886
|
+
if ( ! ( value >= -9223372036854775808.0 && value < 9223372036854775808.0 ) ) {
|
|
887
|
+
carray_jit_whole_real(value, error);
|
|
888
|
+
if ( error && ! isnan(value) && ! isinf(value) ) *error = #{INTEGER_RANGE_CODE};
|
|
889
|
+
return 0;
|
|
890
|
+
}
|
|
891
|
+
return (int64_t) value;
|
|
892
|
+
}
|
|
893
|
+
|
|
894
|
+
C
|
|
895
|
+
end
|
|
896
|
+
if @uses_integer_abs
|
|
897
|
+
text << <<~C
|
|
898
|
+
/* An integer's magnitude, without llabs: that is <stdlib.h>'s,
|
|
899
|
+
which a kernel includes only when it allocates, and it leaves
|
|
900
|
+
INT64_MIN undefined. Negated through uint64_t, INT64_MIN
|
|
901
|
+
wraps to itself -- the wrap an int64 cell has everywhere else. */
|
|
902
|
+
static inline int64_t
|
|
903
|
+
carray_jit_integer_abs (int64_t value)
|
|
904
|
+
{
|
|
905
|
+
return value < 0 ? (int64_t) (0 - (uint64_t) value) : value;
|
|
906
|
+
}
|
|
907
|
+
|
|
908
|
+
C
|
|
909
|
+
end
|
|
680
910
|
COMPLEX_HELPERS.each do |name, definition|
|
|
681
911
|
[false, true].each do |narrow|
|
|
682
912
|
next unless @complex_helpers.include?([name, narrow])
|
|
@@ -684,6 +914,76 @@ class CArray
|
|
|
684
914
|
: definition) << "\n"
|
|
685
915
|
end
|
|
686
916
|
end
|
|
917
|
+
@gamma_types.each do |type|
|
|
918
|
+
suffix, c_type, gamma, modf = GAMMA_C_TYPES.fetch(type)
|
|
919
|
+
table = (1..GAMMA_TABLE_LIMIT).map { |whole|
|
|
920
|
+
format_float(Math.gamma(whole.to_f), type)
|
|
921
|
+
}
|
|
922
|
+
text << <<~C
|
|
923
|
+
/* `Math.gamma` is tgamma with Ruby's two answers around it: a
|
|
924
|
+
whole number up to #{GAMMA_TABLE_LIMIT} is answered from the table Ruby
|
|
925
|
+
answers it from, and a negative whole number -- where tgamma
|
|
926
|
+
gives a NaN -- is the domain error Ruby raises. The values
|
|
927
|
+
are the ones the Ruby that compiled this holds. */
|
|
928
|
+
static const #{c_type} carray_jit_gamma_table_#{suffix}[#{GAMMA_TABLE_LIMIT}] = {
|
|
929
|
+
#{table.each_slice(4).map { |row| row.join(", ") }.join(",\n ")}
|
|
930
|
+
};
|
|
931
|
+
|
|
932
|
+
static inline #{c_type}
|
|
933
|
+
carray_jit_gamma_#{suffix} (#{c_type} x, int32_t *error)
|
|
934
|
+
{
|
|
935
|
+
if ( isinf(x) ) {
|
|
936
|
+
if ( signbit(x) ) {
|
|
937
|
+
if ( error ) *error = 5;
|
|
938
|
+
return 0;
|
|
939
|
+
}
|
|
940
|
+
return x;
|
|
941
|
+
}
|
|
942
|
+
#{c_type} whole;
|
|
943
|
+
if ( #{modf}(x, &whole) == 0 ) {
|
|
944
|
+
if ( whole < 0 ) {
|
|
945
|
+
if ( error ) *error = 5;
|
|
946
|
+
return 0;
|
|
947
|
+
}
|
|
948
|
+
if ( 0 < whole && whole <= #{GAMMA_TABLE_LIMIT} ) {
|
|
949
|
+
return carray_jit_gamma_table_#{suffix}[(int)whole - 1];
|
|
950
|
+
}
|
|
951
|
+
}
|
|
952
|
+
return #{gamma}(x);
|
|
953
|
+
}
|
|
954
|
+
|
|
955
|
+
C
|
|
956
|
+
end
|
|
957
|
+
text << intrinsic_helpers
|
|
958
|
+
@clamp_types.each do |type|
|
|
959
|
+
suffix, c_type, has_nan = CLAMP_C_TYPES.fetch(type)
|
|
960
|
+
refusal = lambda { |test, code|
|
|
961
|
+
" if ( #{test} ) {\n" \
|
|
962
|
+
" if ( error ) *error = #{code};\n" \
|
|
963
|
+
" return value;\n" \
|
|
964
|
+
" }\n"
|
|
965
|
+
}
|
|
966
|
+
tests = +""
|
|
967
|
+
tests << refusal.call("isnan(low) || isnan(high)", 4) if has_nan
|
|
968
|
+
tests << refusal.call("low > high", 3)
|
|
969
|
+
tests << refusal.call("isnan(value)", 4) if has_nan
|
|
970
|
+
text << <<~C
|
|
971
|
+
/* `x.clamp(low, high)`, and the two things Ruby raises
|
|
972
|
+
ArgumentError for: bounds the wrong way round, and a
|
|
973
|
+
comparison that cannot be made because a NaN is in it. The
|
|
974
|
+
order is Ruby's -- the bounds are ordered against each other
|
|
975
|
+
before the value is looked at. */
|
|
976
|
+
static inline #{c_type}
|
|
977
|
+
carray_jit_clamp_#{suffix} (#{c_type} value, #{c_type} low, #{c_type} high, int32_t *error)
|
|
978
|
+
{
|
|
979
|
+
#{tests.chomp}
|
|
980
|
+
if ( value < low ) return low;
|
|
981
|
+
if ( value > high ) return high;
|
|
982
|
+
return value;
|
|
983
|
+
}
|
|
984
|
+
|
|
985
|
+
C
|
|
986
|
+
end
|
|
687
987
|
if @uses_real_arg
|
|
688
988
|
text << <<~C
|
|
689
989
|
/* The argument of a real number, which Ruby reads off the sign
|
|
@@ -757,7 +1057,12 @@ class CArray
|
|
|
757
1057
|
non-zero and disagrees with it in sign -- and, for floats,
|
|
758
1058
|
give a zero remainder the divisor's sign, so the rule holds
|
|
759
1059
|
without exception. This mirrors CArray's own `:mod` kernel in
|
|
760
|
-
`ext/mkkernel.rb`.
|
|
1060
|
+
`ext/mkkernel.rb`.
|
|
1061
|
+
|
|
1062
|
+
A zero divisor is Ruby's rather than CArray's: `3.0 % 0.0` raises
|
|
1063
|
+
ZeroDivisionError in Ruby, a float divisor as much as an integer
|
|
1064
|
+
one, where fmod and CArray answer NaN. `3.0 / 0.0` is not that
|
|
1065
|
+
case -- an infinity in Ruby as in C -- and is left alone. */
|
|
761
1066
|
static inline int64_t
|
|
762
1067
|
carray_jit_floor_modulo (int64_t numerator, int64_t denominator, int32_t *error)
|
|
763
1068
|
{
|
|
@@ -773,8 +1078,12 @@ class CArray
|
|
|
773
1078
|
}
|
|
774
1079
|
|
|
775
1080
|
static inline double
|
|
776
|
-
carray_jit_floor_modulo_real (double numerator, double denominator)
|
|
1081
|
+
carray_jit_floor_modulo_real (double numerator, double denominator, int32_t *error)
|
|
777
1082
|
{
|
|
1083
|
+
if ( denominator == 0 ) {
|
|
1084
|
+
if ( error ) *error = 1;
|
|
1085
|
+
return 0;
|
|
1086
|
+
}
|
|
778
1087
|
double remainder = fmod(numerator, denominator);
|
|
779
1088
|
if ( remainder != 0 ) {
|
|
780
1089
|
if ( (remainder < 0) != (denominator < 0) ) remainder += denominator;
|
|
@@ -814,9 +1123,40 @@ class CArray
|
|
|
814
1123
|
|
|
815
1124
|
C
|
|
816
1125
|
end
|
|
1126
|
+
text << random_source
|
|
817
1127
|
text << pasted_definitions
|
|
818
1128
|
end
|
|
819
1129
|
|
|
1130
|
+
# The generators the block drew from, as CArray wrote them.
|
|
1131
|
+
#
|
|
1132
|
+
# This is the COMPLEX_HELPERS path and not the pasted-body one: what
|
|
1133
|
+
# CArray hands over is a run of `static inline` functions, not one
|
|
1134
|
+
# definition to be given a `static` and a name. It goes in verbatim --
|
|
1135
|
+
# reading it here to check it would be this gem deciding what CArray's
|
|
1136
|
+
# generator is, which is the drift the shared text exists to prevent.
|
|
1137
|
+
#
|
|
1138
|
+
# Only when a draw was written, and once however many generators drew:
|
|
1139
|
+
# two CArray::Rng of the same kind are the same C over different
|
|
1140
|
+
# state.
|
|
1141
|
+
def random_source
|
|
1142
|
+
kinds = @analyzer.random_names.map { |name|
|
|
1143
|
+
@randoms.fetch(name).generator
|
|
1144
|
+
}.uniq.sort
|
|
1145
|
+
return "" if kinds.empty?
|
|
1146
|
+
# What every generator's text wants beside it and none of them owns.
|
|
1147
|
+
# Once, ahead of them: two generators in this file would otherwise
|
|
1148
|
+
# define it twice.
|
|
1149
|
+
text = +"/* What a draw needs beside a generator, from\n" \
|
|
1150
|
+
" CArray::Rng::COMMON_SOURCE. */\n" +
|
|
1151
|
+
CArray::Rng::COMMON_SOURCE + "\n"
|
|
1152
|
+
text + kinds.map { |kind|
|
|
1153
|
+
"/* #{kind}, from CArray::Rng::SOURCE -- the same text CArray\n" \
|
|
1154
|
+
" compiled, so a kernel continues the sequence `random!` left\n" \
|
|
1155
|
+
" off at rather than agreeing with it by construction. */\n" +
|
|
1156
|
+
CArray::Rng::SOURCE.fetch(kind) + "\n"
|
|
1157
|
+
}.join
|
|
1158
|
+
end
|
|
1159
|
+
|
|
820
1160
|
# The bodies of the functions written with `jit_function`, put in this
|
|
821
1161
|
# translation unit as statics and called by name. An address the
|
|
822
1162
|
# compiler cannot see through is a call it cannot inline, and the loop
|
|
@@ -829,18 +1169,49 @@ class CArray
|
|
|
829
1169
|
# one. The name carries the digest of the body, so two of them are the
|
|
830
1170
|
# same function, and one is pasted once however many names the block
|
|
831
1171
|
# reached it by.
|
|
1172
|
+
# Every function whose definition ends up in this file: the ones the
|
|
1173
|
+
# block named, and the ones those call. Helpers and raise messages
|
|
1174
|
+
# are asked of all of them, since a body pasted three deep reports
|
|
1175
|
+
# through the same slot and wants the same preamble.
|
|
1176
|
+
def pasted_closure
|
|
1177
|
+
found = {}
|
|
1178
|
+
walk = lambda do |function|
|
|
1179
|
+
# A borrowed function has no definition to paste; it is declared.
|
|
1180
|
+
next unless function.pasted?
|
|
1181
|
+
next if found[function.symbol]
|
|
1182
|
+
found[function.symbol] = function
|
|
1183
|
+
(function.dependencies || []).each { |called| walk.call(called) }
|
|
1184
|
+
end
|
|
1185
|
+
@pasted_functions.each_value { |function| walk.call(function) }
|
|
1186
|
+
found.values
|
|
1187
|
+
end
|
|
1188
|
+
|
|
1189
|
+
# A pasted body may itself call a function compiled here, which it
|
|
1190
|
+
# reaches by symbol -- so that one is pasted too, and ahead of it, or
|
|
1191
|
+
# the call would be to a name this file has not defined yet. The same
|
|
1192
|
+
# `seen` covers both: one definition per symbol however many ways it
|
|
1193
|
+
# was reached.
|
|
832
1194
|
def pasted_definitions
|
|
833
1195
|
seen = {}
|
|
834
1196
|
text = +""
|
|
835
1197
|
@pasted_functions.each_value do |function|
|
|
836
|
-
|
|
837
|
-
seen[function.name] = true
|
|
838
|
-
text << "/* #{function} */\n" \
|
|
839
|
-
"static #{function.definition}\n"
|
|
1198
|
+
text << paste_definition(function, seen)
|
|
840
1199
|
end
|
|
841
1200
|
text
|
|
842
1201
|
end
|
|
843
1202
|
|
|
1203
|
+
def paste_definition (function, seen)
|
|
1204
|
+
return "" unless function.pasted?
|
|
1205
|
+
return "" if seen[function.symbol]
|
|
1206
|
+
seen[function.symbol] = true
|
|
1207
|
+
text = +""
|
|
1208
|
+
(function.dependencies || []).each do |called|
|
|
1209
|
+
text << paste_definition(called, seen)
|
|
1210
|
+
end
|
|
1211
|
+
text << "/* #{function} */\n" \
|
|
1212
|
+
"static #{function.definition}\n"
|
|
1213
|
+
end
|
|
1214
|
+
|
|
844
1215
|
# One compiled object holds both loops and picks between them once,
|
|
845
1216
|
# outside the loop. The contiguous form indexes a typed pointer along
|
|
846
1217
|
# the innermost axis, which the compiler can vectorise; the strided form
|
|
@@ -848,7 +1219,7 @@ class CArray
|
|
|
848
1219
|
def dispatcher
|
|
849
1220
|
tests = @arrays.map { |array|
|
|
850
1221
|
"strides[#{@stride_offsets.fetch(array) + array_rank(array) - 1}] == " \
|
|
851
|
-
"(int64_t) sizeof(#{storage_c_type(array)})"
|
|
1222
|
+
"(int64_t) sizeof(#{storage_c_type(array_storage(array))})"
|
|
852
1223
|
}
|
|
853
1224
|
# A kernel reaching no array cell has nothing to be contiguous about
|
|
854
1225
|
# -- it can only be one whose work is a call, the arrays it touches
|
|
@@ -878,7 +1249,6 @@ class CArray
|
|
|
878
1249
|
|
|
879
1250
|
def build_body (contiguous)
|
|
880
1251
|
@contiguous = contiguous
|
|
881
|
-
@declared_locals = {}
|
|
882
1252
|
@temporary_count = 0
|
|
883
1253
|
@carried_masks = []
|
|
884
1254
|
# Whether this body has a way to report -- asked of the statements
|
|
@@ -886,10 +1256,202 @@ class CArray
|
|
|
886
1256
|
# kernel carries every extent of its arrays so the border rule has
|
|
887
1257
|
# them, which says nothing about whether a cell can fail.
|
|
888
1258
|
@body_reports = false
|
|
1259
|
+
indent = " " + " " * @rank
|
|
889
1260
|
statements = @analyzer.body.statements.map { |statement|
|
|
890
|
-
emit_statement(statement,
|
|
1261
|
+
emit_statement(statement, indent)
|
|
1262
|
+
}.join
|
|
1263
|
+
"{\n" + declarations + heap_prologue + loop_open +
|
|
1264
|
+
local_declarations(@analyzer.body, indent) + statements +
|
|
1265
|
+
loop_close + heap_epilogue + "}\n"
|
|
1266
|
+
end
|
|
1267
|
+
|
|
1268
|
+
# The arrays this kernel allocates, in the order the block declared
|
|
1269
|
+
# them, as [C name, storage, shape].
|
|
1270
|
+
#
|
|
1271
|
+
# Gathered from every scope rather than from the one being emitted: the
|
|
1272
|
+
# allocation is at the function's entry, outside the cell loop, so a
|
|
1273
|
+
# pointer declared for an array the block made inside an inner loop
|
|
1274
|
+
# still stands at the top of the function. Two sibling loops that each
|
|
1275
|
+
# write `w` are two bindings and so two pointers, which is what keeps
|
|
1276
|
+
# them the two arrays Ruby says they are.
|
|
1277
|
+
def heap_arrays
|
|
1278
|
+
@heap_arrays ||= begin
|
|
1279
|
+
found = []
|
|
1280
|
+
walk = lambda do |node|
|
|
1281
|
+
if node.respond_to?(:array_declarations) && node.array_declarations
|
|
1282
|
+
node.array_declarations.each do |name, binding, storage, shape, heap|
|
|
1283
|
+
next unless heap
|
|
1284
|
+
entry = [local_c_name(name, binding), storage, shape]
|
|
1285
|
+
found << entry unless found.include?(entry)
|
|
1286
|
+
end
|
|
1287
|
+
end
|
|
1288
|
+
node.children.each { |child| walk.call(child) if child.is_a?(Node) }
|
|
1289
|
+
end
|
|
1290
|
+
walk.call(@analyzer.body)
|
|
1291
|
+
found
|
|
1292
|
+
end
|
|
1293
|
+
end
|
|
1294
|
+
|
|
1295
|
+
# One allocation per array at the entry, before the loop: the cells,
|
|
1296
|
+
# and the shadow beside them where the kernel carries masks.
|
|
1297
|
+
#
|
|
1298
|
+
# The lengths come first and stand in `const int64_t` of their own,
|
|
1299
|
+
# because each is read three times over -- to check it, to size the
|
|
1300
|
+
# allocation, and at every subscript on that axis -- and the expression
|
|
1301
|
+
# behind it may be anything the block wrote. A length that is not a
|
|
1302
|
+
# whole number of cells above zero is reported rather than allocated,
|
|
1303
|
+
# which is what the same shape written as a number is refused for as
|
|
1304
|
+
# the block is read. So is a length whose cells would not fit in a
|
|
1305
|
+
# `size_t`: the multiplication into `malloc` would wrap, and a small
|
|
1306
|
+
# allocation answered for a huge one is the worst of the outcomes.
|
|
1307
|
+
#
|
|
1308
|
+
# That is asked of the lengths before they are multiplied, one axis at
|
|
1309
|
+
# a time against what the axes before it leave of the limit. Asked of
|
|
1310
|
+
# the product, it came too late: two lengths of 2**32 multiply to a
|
|
1311
|
+
# number that has already wrapped to zero, which fits, and every
|
|
1312
|
+
# subscript was then checked against its own axis and let through
|
|
1313
|
+
# into an allocation of nothing. The limit is `int64_t` as well as
|
|
1314
|
+
# `size_t`, since that is what the cells are counted in.
|
|
1315
|
+
def heap_prologue
|
|
1316
|
+
return "" if heap_arrays.empty?
|
|
1317
|
+
lines = +""
|
|
1318
|
+
heap_arrays.each do |variable, storage, shape|
|
|
1319
|
+
shape.each_with_index do |extent, axis|
|
|
1320
|
+
next if extent.is_a?(Integer)
|
|
1321
|
+
lines << " const int64_t #{heap_extent_name(variable, axis)} = " \
|
|
1322
|
+
"#{emit(extent.node, :int64)};\n"
|
|
1323
|
+
end
|
|
1324
|
+
room = "(uint64_t) (SIZE_MAX / sizeof(#{storage_c_type(storage)}))"
|
|
1325
|
+
lines << " const uint64_t #{heap_limit_name(variable)} =\n" \
|
|
1326
|
+
" #{room} < (uint64_t) INT64_MAX ? #{room} : (uint64_t) INT64_MAX;\n"
|
|
1327
|
+
lengths = heap_extent_texts(variable, shape)
|
|
1328
|
+
tests = shape.each_with_index.filter_map { |extent, axis|
|
|
1329
|
+
next if extent.is_a?(Integer)
|
|
1330
|
+
"#{heap_extent_name(variable, axis)} >= 1"
|
|
1331
|
+
} + lengths.each_index.map { |axis|
|
|
1332
|
+
left = [heap_limit_name(variable)] +
|
|
1333
|
+
lengths.first(axis).map { |length| "(uint64_t) #{length}" }
|
|
1334
|
+
"(uint64_t) #{lengths[axis]} <= #{left.join(' / ')}"
|
|
1335
|
+
}
|
|
1336
|
+
lines << " const int #{heap_fits_name(variable)} =\n" \
|
|
1337
|
+
" #{tests.join("\n && ")};\n"
|
|
1338
|
+
next if literal_shape?(shape)
|
|
1339
|
+
lines << " const int64_t #{heap_cells_name(variable)} = " \
|
|
1340
|
+
"#{heap_fits_name(variable)} ? #{lengths.join(' * ')} : 0;\n"
|
|
1341
|
+
end
|
|
1342
|
+
heap_arrays.each do |variable, storage, _shape|
|
|
1343
|
+
lines << " #{storage_c_type(storage)} *#{variable} = NULL;\n"
|
|
1344
|
+
lines << " uint8_t *#{local_mask_name(variable)} = NULL;\n" if @masked
|
|
1345
|
+
end
|
|
1346
|
+
unfit = heap_arrays.map { |variable, _storage, _shape|
|
|
1347
|
+
"!#{heap_fits_name(variable)}"
|
|
1348
|
+
}
|
|
1349
|
+
lines << " if ( #{unfit.join(' || ')} ) {\n" \
|
|
1350
|
+
" if ( error ) *error = #{SHAPE_CODE};\n" \
|
|
1351
|
+
" return;\n" \
|
|
1352
|
+
" }\n"
|
|
1353
|
+
heap_arrays.each do |variable, storage, shape|
|
|
1354
|
+
cells = local_array_cells_text(variable, shape)
|
|
1355
|
+
lines << " #{variable} = malloc(sizeof(#{storage_c_type(storage)}) " \
|
|
1356
|
+
"* (size_t) #{cells});\n"
|
|
1357
|
+
next unless @masked
|
|
1358
|
+
lines << " #{local_mask_name(variable)} = malloc((size_t) #{cells});\n"
|
|
1359
|
+
end
|
|
1360
|
+
held = heap_arrays.flat_map { |variable, _storage, _shape|
|
|
1361
|
+
[variable] + (@masked ? [local_mask_name(variable)] : [])
|
|
1362
|
+
}
|
|
1363
|
+
lines << " if ( #{held.map { |name| "!#{name}" }.join(' || ')} ) {\n"
|
|
1364
|
+
held.each { |name| lines << " free(#{name});\n" }
|
|
1365
|
+
lines << " if ( error ) *error = #{MEMORY_CODE};\n" \
|
|
1366
|
+
" return;\n" \
|
|
1367
|
+
" }\n\n"
|
|
1368
|
+
lines
|
|
1369
|
+
end
|
|
1370
|
+
|
|
1371
|
+
# Every way out of a kernel that allocated: the fall-through at the end
|
|
1372
|
+
# of the loop, and the one `return` a `raise` emits, which stands
|
|
1373
|
+
# inside it. Counted rather than assumed -- a leak here is a leak per
|
|
1374
|
+
# call, and the sweep calls once per chunk.
|
|
1375
|
+
def heap_epilogue
|
|
1376
|
+
heap_release(" ")
|
|
1377
|
+
end
|
|
1378
|
+
|
|
1379
|
+
def heap_release (indent)
|
|
1380
|
+
return "" if heap_arrays.empty?
|
|
1381
|
+
heap_arrays.map { |variable, _storage, _shape|
|
|
1382
|
+
"#{indent}free(#{variable});\n" +
|
|
1383
|
+
(@masked ? "#{indent}free(#{local_mask_name(variable)});\n" : "")
|
|
1384
|
+
}.join
|
|
1385
|
+
end
|
|
1386
|
+
|
|
1387
|
+
def heap_extent_name (variable, axis)
|
|
1388
|
+
"#{variable}__extent#{axis}"
|
|
1389
|
+
end
|
|
1390
|
+
|
|
1391
|
+
def heap_cells_name (variable)
|
|
1392
|
+
"#{variable}__cells"
|
|
1393
|
+
end
|
|
1394
|
+
|
|
1395
|
+
def heap_limit_name (variable)
|
|
1396
|
+
"#{variable}__limit"
|
|
1397
|
+
end
|
|
1398
|
+
|
|
1399
|
+
def heap_fits_name (variable)
|
|
1400
|
+
"#{variable}__fits"
|
|
1401
|
+
end
|
|
1402
|
+
|
|
1403
|
+
def heap_extent_texts (variable, shape)
|
|
1404
|
+
shape.each_with_index.map { |extent, axis|
|
|
1405
|
+
extent.is_a?(Integer) ? extent.to_s : heap_extent_name(variable, axis)
|
|
1406
|
+
}
|
|
1407
|
+
end
|
|
1408
|
+
|
|
1409
|
+
def literal_shape? (shape)
|
|
1410
|
+
shape.all? { |extent| extent.is_a?(Integer) }
|
|
1411
|
+
end
|
|
1412
|
+
|
|
1413
|
+
# The locals a scope holds, declared at the head of its block with no
|
|
1414
|
+
# value: every assignment below is a plain one, wherever it stands --
|
|
1415
|
+
# inside a `while` or an arm of an `if` included -- and nothing is read
|
|
1416
|
+
# before it is written, which the typing has already made sure of.
|
|
1417
|
+
def local_declarations (scope, indent)
|
|
1418
|
+
scope.declarations.map { |name, binding, type|
|
|
1419
|
+
variable = local_c_name(name, binding)
|
|
1420
|
+
"#{indent}#{COMPUTATION_C_TYPES.fetch(type)} #{variable};\n" +
|
|
1421
|
+
(@masked ? "#{indent}uint8_t #{local_mask_name(variable)};\n" : "")
|
|
1422
|
+
}.join + local_array_declarations(scope, indent)
|
|
1423
|
+
end
|
|
1424
|
+
|
|
1425
|
+
# The arrays a scope holds, declared at the head of its block beside the
|
|
1426
|
+
# locals. The length is the one the block wrote, so the C array is a
|
|
1427
|
+
# real one -- a fixed-size automatic object, not a pointer into
|
|
1428
|
+
# anything -- and its cells are addressed by constants.
|
|
1429
|
+
#
|
|
1430
|
+
# What clears the zeroed ones is emitted where the statement stands
|
|
1431
|
+
# rather than here: in Ruby a fresh array is made at that line each time
|
|
1432
|
+
# it runs, so a loop's body starts from cleared cells on every pass.
|
|
1433
|
+
def local_array_declarations (scope, indent)
|
|
1434
|
+
scope.array_declarations.map { |name, binding, storage, shape, heap|
|
|
1435
|
+
# One the kernel allocates is a pointer standing at the top of the
|
|
1436
|
+
# function, declared and freed with the allocation (see
|
|
1437
|
+
# #heap_prologue), so the scope it belongs to declares nothing.
|
|
1438
|
+
next "" if heap
|
|
1439
|
+
# One flat run of cells, whatever the rank: the subscripts are
|
|
1440
|
+
# folded to a row-major offset with the strides baked in (see
|
|
1441
|
+
# #local_array_reference), and a flat array is also what a C
|
|
1442
|
+
# function's pointer parameter is handed. `double m[3][4]` would
|
|
1443
|
+
# be the same bytes and a second spelling to keep in step.
|
|
1444
|
+
variable = local_c_name(name, binding)
|
|
1445
|
+
cells = shape.inject(1, :*)
|
|
1446
|
+
# Where the kernel carries masks, a shadow of one byte a cell
|
|
1447
|
+
# beside the cells -- the same arrangement a scalar local has, one
|
|
1448
|
+
# byte beside its value, spread over the cells. A cell of a local
|
|
1449
|
+
# array then reads like a cell of a captured one: the value, and
|
|
1450
|
+
# whether it is there.
|
|
1451
|
+
"#{indent}#{storage_c_type(storage)} #{variable}[#{cells}];\n" +
|
|
1452
|
+
(@masked ?
|
|
1453
|
+
"#{indent}uint8_t #{local_mask_name(variable)}[#{cells}];\n" : "")
|
|
891
1454
|
}.join
|
|
892
|
-
"{\n" + declarations + loop_open + statements + loop_close + "}\n"
|
|
893
1455
|
end
|
|
894
1456
|
|
|
895
1457
|
def array_rank (array)
|
|
@@ -930,30 +1492,38 @@ class CArray
|
|
|
930
1492
|
end
|
|
931
1493
|
end
|
|
932
1494
|
@reals.each_with_index do |name, index|
|
|
933
|
-
lines << " const double #{
|
|
1495
|
+
lines << " const double #{bare_name(name)} = reals[#{index}];\n"
|
|
934
1496
|
end
|
|
935
1497
|
@complexes.each_with_index do |name, index|
|
|
936
1498
|
slot = @reals.size + 2 * index
|
|
937
|
-
lines << " const double _Complex #{
|
|
1499
|
+
lines << " const double _Complex #{bare_name(name)} = " \
|
|
938
1500
|
"CMPLX(reals[#{slot}], reals[#{slot + 1}]);\n"
|
|
939
1501
|
end
|
|
940
1502
|
@integers.each_with_index do |name, index|
|
|
941
|
-
lines << " const int64_t #{
|
|
1503
|
+
lines << " const int64_t #{bare_name(name)} = integers[#{index}];\n"
|
|
1504
|
+
end
|
|
1505
|
+
# The cast is the whole of how a uint64 capture travels: the caller
|
|
1506
|
+
# packed the value's bits into the slot, and reading them as what they
|
|
1507
|
+
# are is what puts the value back.
|
|
1508
|
+
@unsigned_integers.each_with_index do |name, index|
|
|
1509
|
+
lines << " const uint64_t #{bare_name(name)} = " \
|
|
1510
|
+
"(uint64_t) integers[#{@integers.size + index}];\n"
|
|
942
1511
|
end
|
|
943
1512
|
@extent_slots.each_with_index do |(array, axis), slot|
|
|
944
1513
|
lines << " const int64_t #{extent_name(array, axis)} = " \
|
|
945
|
-
"integers[#{@integers.size + slot}];\n"
|
|
1514
|
+
"integers[#{@integers.size + @unsigned_integers.size + slot}];\n"
|
|
946
1515
|
end
|
|
947
1516
|
@address_functions.each_key.with_index do |name, index|
|
|
948
|
-
lines << " const #{c_function_type_name(name)} #{
|
|
1517
|
+
lines << " const #{c_function_type_name(name)} #{bare_name(name)} = " \
|
|
949
1518
|
"(#{c_function_type_name(name)}) functions[#{index}];\n"
|
|
950
1519
|
end
|
|
951
1520
|
@address_arrays.each_with_index do |array, index|
|
|
952
1521
|
# Typed by the array rather than by the parameter it will be passed
|
|
953
1522
|
# to: the caller has already checked that the two agree, and the
|
|
954
1523
|
# array is the one that owns the memory.
|
|
955
|
-
|
|
956
|
-
|
|
1524
|
+
storage = storage_c_type(array_storage(array))
|
|
1525
|
+
lines << " #{storage} *const #{bare_name(array)} = " \
|
|
1526
|
+
"(#{storage} *) data[#{index}];\n"
|
|
957
1527
|
end
|
|
958
1528
|
lines.empty? ? "" : lines.join + "\n"
|
|
959
1529
|
end
|
|
@@ -997,6 +1567,20 @@ class CArray
|
|
|
997
1567
|
def emit_statement (statement, indent)
|
|
998
1568
|
case statement
|
|
999
1569
|
when Assignment then emit_assignment(statement, indent)
|
|
1570
|
+
when ParallelAssignment then
|
|
1571
|
+
# The values, and then the writes. They are ordinary statements by
|
|
1572
|
+
# the time they reach here -- the analyzer made them so -- and the
|
|
1573
|
+
# order they are in is the whole of what a parallel assignment
|
|
1574
|
+
# means.
|
|
1575
|
+
statement.statements.map { |inner|
|
|
1576
|
+
emit_statement(inner, indent)
|
|
1577
|
+
}.join
|
|
1578
|
+
when IntrinsicStatement then emit_intrinsic_statement(statement, indent)
|
|
1579
|
+
when LocalArrayDeclaration then emit_local_array_clearing(statement, indent)
|
|
1580
|
+
when LocalArrayWrite then guarded(statement, indent) { |inner|
|
|
1581
|
+
emit_local_array_write(statement, inner) }
|
|
1582
|
+
when LocalArrayMaskWrite then guarded(statement, indent) { |inner|
|
|
1583
|
+
emit_local_array_mask_write(statement, inner) }
|
|
1000
1584
|
when ElementWrite then guarded(statement, indent) { |inner|
|
|
1001
1585
|
emit_element_write(statement, inner) }
|
|
1002
1586
|
when MaskWrite then guarded(statement, indent) { |inner|
|
|
@@ -1019,6 +1603,33 @@ class CArray
|
|
|
1019
1603
|
|
|
1020
1604
|
# 1 and 2 are the divisor that was not there and the subscript that ran
|
|
1021
1605
|
# off its array. A `raise` in the block takes a code from here up.
|
|
1606
|
+
# What a kernel says when it could not start: a shape it worked out
|
|
1607
|
+
# that is no shape, and an allocation that failed. Negative, because
|
|
1608
|
+
# the positive codes from RAISE_CODE_FLOOR up are a digest of a
|
|
1609
|
+
# `raise` message and a fixed one among them could collide; nothing
|
|
1610
|
+
# reaches for a negative code, so the fixed ones are kept down here,
|
|
1611
|
+
# the kernel's own.
|
|
1612
|
+
SHAPE_CODE = -1
|
|
1613
|
+
MEMORY_CODE = -2
|
|
1614
|
+
|
|
1615
|
+
# And what a rounding reports where Ruby has no Integer to answer, or
|
|
1616
|
+
# has one an int64 cannot hold. Negative for the same reason: fixed
|
|
1617
|
+
# codes, which a `raise` digest must not land on.
|
|
1618
|
+
FLOAT_NAN_CODE = -3
|
|
1619
|
+
FLOAT_INFINITY_CODE = -4
|
|
1620
|
+
FLOAT_NEGATIVE_INFINITY_CODE = -5
|
|
1621
|
+
INTEGER_RANGE_CODE = -6
|
|
1622
|
+
|
|
1623
|
+
# The fixed failures a body reports, as the exception each one raises.
|
|
1624
|
+
# A kernel and a compiled function both look here, so the two say the
|
|
1625
|
+
# same thing for the same cell.
|
|
1626
|
+
FIXED_FAILURES = {
|
|
1627
|
+
FLOAT_NAN_CODE => [FloatDomainError, "NaN"],
|
|
1628
|
+
FLOAT_INFINITY_CODE => [FloatDomainError, "Infinity"],
|
|
1629
|
+
FLOAT_NEGATIVE_INFINITY_CODE => [FloatDomainError, "-Infinity"],
|
|
1630
|
+
INTEGER_RANGE_CODE => [RangeError, "a rounded Float is past the range of int64"],
|
|
1631
|
+
}.freeze
|
|
1632
|
+
|
|
1022
1633
|
RAISE_CODE_FLOOR = 3
|
|
1023
1634
|
|
|
1024
1635
|
# The code is taken from the message rather than counted off as messages
|
|
@@ -1070,7 +1681,8 @@ class CArray
|
|
|
1070
1681
|
# The value is not the answer; the flag says so.
|
|
1071
1682
|
leaving = @in_function && !@returns_nothing ? "return 0;" : "return;"
|
|
1072
1683
|
"#{indent}if ( #{tests.join(' && ')} ) {\n" \
|
|
1073
|
-
"#{indent} #{report}\n"
|
|
1684
|
+
"#{indent} #{report}\n" +
|
|
1685
|
+
heap_release("#{indent} ") +
|
|
1074
1686
|
"#{indent} #{leaving}\n" \
|
|
1075
1687
|
"#{indent}}\n"
|
|
1076
1688
|
end
|
|
@@ -1229,17 +1841,28 @@ class CArray
|
|
|
1229
1841
|
# report -- but writing outside it cannot: the report would arrive after
|
|
1230
1842
|
# the damage. So a scatter works its positions out first, and writes
|
|
1231
1843
|
# only if every one of them is inside.
|
|
1844
|
+
#
|
|
1845
|
+
# Each position carries the text of the extent it is tested against
|
|
1846
|
+
# rather than an array and an axis to look one up from. An operand's
|
|
1847
|
+
# extent is a variable the kernel was handed, and there are extents that
|
|
1848
|
+
# are not one of those; what the test needs is the text either way.
|
|
1232
1849
|
def guarded (write, indent)
|
|
1233
1850
|
positions = scattered_positions(write)
|
|
1234
1851
|
return yield(indent) if positions.empty?
|
|
1235
1852
|
|
|
1236
1853
|
lines = "#{indent}{\n"
|
|
1237
|
-
|
|
1854
|
+
# Held by axis as well as by node: an axis whose length the kernel
|
|
1855
|
+
# works out is tested here even where the position is an index, and
|
|
1856
|
+
# an index is not a node the reference can look itself up by.
|
|
1857
|
+
held = {}
|
|
1858
|
+
positions.each_with_index do |(temporary, node, _extent), axis|
|
|
1238
1859
|
lines << "#{indent} const int64_t #{temporary} = #{emit(node, :int64)};\n"
|
|
1239
1860
|
@position_temporaries[node] = temporary
|
|
1861
|
+
held[positions_axis(write, axis)] = temporary
|
|
1240
1862
|
end
|
|
1241
|
-
|
|
1242
|
-
|
|
1863
|
+
@guarded_axes[write] = held
|
|
1864
|
+
test = positions.map { |temporary, _node, extent|
|
|
1865
|
+
"#{temporary} >= 0 && #{temporary} < #{extent}"
|
|
1243
1866
|
}.join(" && ")
|
|
1244
1867
|
lines << "#{indent} if ( #{test} ) {\n"
|
|
1245
1868
|
lines << yield("#{indent} ")
|
|
@@ -1247,19 +1870,74 @@ class CArray
|
|
|
1247
1870
|
lines << "#{indent} if ( #{error_argument} ) *#{error_argument} = 2;\n"
|
|
1248
1871
|
lines << "#{indent} }\n"
|
|
1249
1872
|
lines << "#{indent}}\n"
|
|
1250
|
-
positions.each { |_, node, _
|
|
1873
|
+
positions.each { |_, node, _| @position_temporaries.delete(node) }
|
|
1874
|
+
@guarded_axes.delete(write)
|
|
1251
1875
|
lines
|
|
1252
1876
|
end
|
|
1253
1877
|
|
|
1878
|
+
# Which axis the nth guarded position belongs to.
|
|
1879
|
+
def positions_axis (write, nth)
|
|
1880
|
+
guarded_axes_of(write).fetch(nth)
|
|
1881
|
+
end
|
|
1882
|
+
|
|
1883
|
+
def guarded_axes_of (write)
|
|
1884
|
+
write_subscripts(write).each_with_index.filter_map { |(index, offset), axis|
|
|
1885
|
+
axis if guarded_axis?(write, index, offset, axis)
|
|
1886
|
+
}
|
|
1887
|
+
end
|
|
1888
|
+
|
|
1889
|
+
# [temporary, position, extent text] for each axis of a write that is
|
|
1890
|
+
# addressed at a position only the running kernel knows.
|
|
1254
1891
|
def scattered_positions (write)
|
|
1255
1892
|
subscripts = write_subscripts(write)
|
|
1256
1893
|
subscripts.each_with_index.filter_map { |(index, offset), axis|
|
|
1257
|
-
|
|
1258
|
-
|
|
1259
|
-
|
|
1894
|
+
# An axis whose length the kernel works out has nothing that was
|
|
1895
|
+
# settled as the block was read -- a number, an index, an index
|
|
1896
|
+
# with an offset, all of them -- so every write onto one works its
|
|
1897
|
+
# position out here and is tested before it lands.
|
|
1898
|
+
next unless guarded_axis?(write, index, offset, axis)
|
|
1899
|
+
position = if index.nil? && offset.is_a?(Node) then offset
|
|
1900
|
+
elsif index.nil? then IntegerLiteral.new(offset)
|
|
1901
|
+
elsif offset.is_a?(Node) then offset
|
|
1902
|
+
else index_position_node(index, offset)
|
|
1903
|
+
end
|
|
1904
|
+
[next_temporary("position"), position, write_extent(write, axis)]
|
|
1260
1905
|
}
|
|
1261
1906
|
end
|
|
1262
1907
|
|
|
1908
|
+
# `k`, `k + 1`: the position an index and its offset stand for, as a
|
|
1909
|
+
# node, so that a write onto an axis the kernel sized can work it out
|
|
1910
|
+
# into a temporary and test it the way a scatter does.
|
|
1911
|
+
def index_position_node (index, offset)
|
|
1912
|
+
read = IndexVariable.new(index, 0)
|
|
1913
|
+
read.type = :int64
|
|
1914
|
+
return read if offset.zero?
|
|
1915
|
+
built = BinaryOperation.new(offset.negative? ? :- : :+, read,
|
|
1916
|
+
IntegerLiteral.new(offset.abs))
|
|
1917
|
+
built.type = :int64
|
|
1918
|
+
built
|
|
1919
|
+
end
|
|
1920
|
+
|
|
1921
|
+
def sized_at_runtime? (write, axis)
|
|
1922
|
+
write.respond_to?(:shape) && !write.shape.fetch(axis).is_a?(Integer)
|
|
1923
|
+
end
|
|
1924
|
+
|
|
1925
|
+
def guarded_axis? (write, index, offset, axis)
|
|
1926
|
+
return true if sized_at_runtime?(write, axis)
|
|
1927
|
+
index.nil? && offset.is_a?(Node) && !Analyzer.fixed_subscript?(offset)
|
|
1928
|
+
end
|
|
1929
|
+
|
|
1930
|
+
# The extent an axis of a write is tested against. An operand's is a
|
|
1931
|
+
# variable the kernel was handed; an array the body made carries the
|
|
1932
|
+
# length the block wrote, and there is no variable to name.
|
|
1933
|
+
def write_extent (write, axis)
|
|
1934
|
+
case write
|
|
1935
|
+
when LocalArrayWrite, LocalArrayMaskWrite
|
|
1936
|
+
local_array_extent_text(write, axis)
|
|
1937
|
+
else extent_name(write.array, axis)
|
|
1938
|
+
end
|
|
1939
|
+
end
|
|
1940
|
+
|
|
1263
1941
|
# `out[i] = UNDEF` marks the cell missing. The bytes underneath are out
|
|
1264
1942
|
# of contract, so there is nothing to store into them.
|
|
1265
1943
|
def emit_mask_only_write (write, indent)
|
|
@@ -1267,8 +1945,23 @@ class CArray
|
|
|
1267
1945
|
end
|
|
1268
1946
|
|
|
1269
1947
|
# The loop a reduction runs in, inside one cell of the outer loops.
|
|
1948
|
+
#
|
|
1949
|
+
# A failure that reports and carries on -- a division handed back a
|
|
1950
|
+
# zero, a computed index handed back cell 0 -- is stopped here, at the
|
|
1951
|
+
# head of the next pass, in a function as in a kernel. The flag is
|
|
1952
|
+
# spelled three ways: a kernel's slot, the flag standing in a function's
|
|
1953
|
+
# own object, and the pointer a pasted body is handed, which a kernel
|
|
1954
|
+
# passes as null under a masked cell and so is asked about first.
|
|
1955
|
+
#
|
|
1956
|
+
# Called once the loop's body is emitted, since the body is what says
|
|
1957
|
+
# whether there is anything to report.
|
|
1270
1958
|
def inner_loop_guard (indent)
|
|
1271
|
-
|
|
1959
|
+
return "" unless @body_reports
|
|
1960
|
+
flag = if !@in_function then "*error"
|
|
1961
|
+
elsif @error_parameter then "#{ERROR_FLAG} && *#{ERROR_FLAG}"
|
|
1962
|
+
else ERROR_FLAG
|
|
1963
|
+
end
|
|
1964
|
+
"#{indent} if ( #{flag} ) break;\n"
|
|
1272
1965
|
end
|
|
1273
1966
|
|
|
1274
1967
|
# The same shape the bounded loop gets, minus the counter -- including
|
|
@@ -1294,13 +1987,13 @@ class CArray
|
|
|
1294
1987
|
@carried_masks =
|
|
1295
1988
|
@masked ? (outer + [emit_mask(loop_node.condition)]).uniq : outer
|
|
1296
1989
|
|
|
1297
|
-
|
|
1298
|
-
|
|
1299
|
-
text += loop_node.statements.map { |statement|
|
|
1990
|
+
condition = emit(loop_node.condition, :boolean)
|
|
1991
|
+
body = loop_node.statements.map { |statement|
|
|
1300
1992
|
emit_statement(statement, indent + " ")
|
|
1301
1993
|
}.join
|
|
1302
1994
|
@carried_masks = outer
|
|
1303
|
-
|
|
1995
|
+
"#{indent}while (#{condition}) {\n" + inner_loop_guard(indent) +
|
|
1996
|
+
body + "#{indent}}\n"
|
|
1304
1997
|
end
|
|
1305
1998
|
|
|
1306
1999
|
def emit_inner_loop (loop_node, indent)
|
|
@@ -1308,13 +2001,15 @@ class CArray
|
|
|
1308
2001
|
return emit_split_reduction(loop_node, accumulation, indent) if accumulation
|
|
1309
2002
|
|
|
1310
2003
|
index = loop_node.index
|
|
1311
|
-
|
|
1312
|
-
|
|
1313
|
-
|
|
1314
|
-
|
|
1315
|
-
|
|
1316
|
-
|
|
1317
|
-
|
|
2004
|
+
step = loop_node.step
|
|
2005
|
+
opening = "#{indent}for (int64_t #{index} = #{emit(loop_node.from, :int64)}; " \
|
|
2006
|
+
"#{index} #{step.positive? ? '<' : '>'} " \
|
|
2007
|
+
"#{emit(loop_node.to, :int64)}; #{stride_step(index, step)}) {\n"
|
|
2008
|
+
body = local_declarations(loop_node, indent + " ") +
|
|
2009
|
+
loop_node.statements.map { |statement|
|
|
2010
|
+
emit_statement(statement, indent + " ")
|
|
2011
|
+
}.join
|
|
2012
|
+
opening + inner_loop_guard(indent) + body + "#{indent}}\n"
|
|
1318
2013
|
end
|
|
1319
2014
|
|
|
1320
2015
|
# The loop is a reduction when its whole body is one local folding a
|
|
@@ -1329,14 +2024,20 @@ class CArray
|
|
|
1329
2024
|
# A masked accumulator carries a mask beside its value, and a partial
|
|
1330
2025
|
# sum would need one each. Left out rather than half-done.
|
|
1331
2026
|
return nil if @masked
|
|
2027
|
+
# The split walks the index a round at a time from one end, so it is
|
|
2028
|
+
# written for a loop that counts by one. A stride or a downward
|
|
2029
|
+
# sweep keeps the serial loop -- which costs nothing that was there
|
|
2030
|
+
# before, those loops having been unwritable until now.
|
|
2031
|
+
return nil unless loop_node.step == 1
|
|
1332
2032
|
return nil unless loop_node.statements.size == 1
|
|
1333
2033
|
|
|
1334
2034
|
assignment = loop_node.statements.first
|
|
1335
2035
|
return nil unless assignment.is_a?(Assignment)
|
|
1336
|
-
|
|
1337
|
-
#
|
|
1338
|
-
# value rather than starting one.
|
|
1339
|
-
return nil unless
|
|
2036
|
+
local = assignment.local
|
|
2037
|
+
# Live when this loop was entered, at this type and binding, so the
|
|
2038
|
+
# fold continues a value rather than starting one.
|
|
2039
|
+
return nil unless loop_node.entering&.[](assignment.name) ==
|
|
2040
|
+
[assignment.type, assignment.binding]
|
|
1340
2041
|
|
|
1341
2042
|
expression = assignment.expression
|
|
1342
2043
|
return nil unless expression.is_a?(BinaryOperation)
|
|
@@ -1351,9 +2052,9 @@ class CArray
|
|
|
1351
2052
|
return nil if literal_extent(loop_node)&.<(PARTIAL_ACCUMULATORS)
|
|
1352
2053
|
|
|
1353
2054
|
left, right = expression.left, expression.right
|
|
1354
|
-
if accumulator?(left,
|
|
2055
|
+
if accumulator?(left, local) && !mentions_local?(right, local)
|
|
1355
2056
|
[assignment, operator]
|
|
1356
|
-
elsif accumulator?(right,
|
|
2057
|
+
elsif accumulator?(right, local) && !mentions_local?(left, local)
|
|
1357
2058
|
[assignment, operator]
|
|
1358
2059
|
end
|
|
1359
2060
|
end
|
|
@@ -1376,13 +2077,14 @@ class CArray
|
|
|
1376
2077
|
end
|
|
1377
2078
|
end
|
|
1378
2079
|
|
|
1379
|
-
|
|
1380
|
-
|
|
2080
|
+
# `local` is [name, binding]: the variable, not how it is spelled.
|
|
2081
|
+
def accumulator? (node, local)
|
|
2082
|
+
node.is_a?(LocalRead) && node.local == local
|
|
1381
2083
|
end
|
|
1382
2084
|
|
|
1383
|
-
def mentions_local? (node,
|
|
1384
|
-
return true if accumulator?(node,
|
|
1385
|
-
node.children.any? { |child| mentions_local?(child,
|
|
2085
|
+
def mentions_local? (node, local)
|
|
2086
|
+
return true if accumulator?(node, local)
|
|
2087
|
+
node.children.any? { |child| mentions_local?(child, local) }
|
|
1386
2088
|
end
|
|
1387
2089
|
|
|
1388
2090
|
# The fold, run as PARTIAL_ACCUMULATORS chains instead of one, with the
|
|
@@ -1392,14 +2094,27 @@ class CArray
|
|
|
1392
2094
|
# the operator's identity, so it is folded in exactly once. Each chain
|
|
1393
2095
|
# keeps the term's own operand order -- what is licensed here is the
|
|
1394
2096
|
# order the *iterations* are grouped in, not the order within one.
|
|
2097
|
+
# `k++`, `k--`, or the stride written out.
|
|
2098
|
+
def stride_step (index, step)
|
|
2099
|
+
case step
|
|
2100
|
+
when 1 then "#{index}++"
|
|
2101
|
+
when -1 then "#{index}--"
|
|
2102
|
+
else step.positive? ? "#{index} += #{step}" : "#{index} -= #{-step}"
|
|
2103
|
+
end
|
|
2104
|
+
end
|
|
2105
|
+
|
|
1395
2106
|
def emit_split_reduction (loop_node, accumulation, indent)
|
|
1396
2107
|
assignment, operator = accumulation
|
|
1397
|
-
name = assignment.
|
|
2108
|
+
name = local_c_name(assignment.name, assignment.binding)
|
|
1398
2109
|
type = assignment.type
|
|
1399
2110
|
index = loop_node.index
|
|
1400
2111
|
base = "#{index}__base"
|
|
1401
2112
|
limit = "#{index}__end"
|
|
1402
|
-
|
|
2113
|
+
# A lane belongs to the accumulator, so its C name comes the way every
|
|
2114
|
+
# local's does.
|
|
2115
|
+
lanes = (0...PARTIAL_ACCUMULATORS).map { |lane|
|
|
2116
|
+
local_c_name(assignment.name, assignment.binding, lane)
|
|
2117
|
+
}
|
|
1403
2118
|
identity = (operator == :* ? ONES : ZEROES).fetch(type)
|
|
1404
2119
|
|
|
1405
2120
|
inner = indent + " "
|
|
@@ -1408,22 +2123,22 @@ class CArray
|
|
|
1408
2123
|
text += "#{inner}const int64_t #{limit} = #{emit(loop_node.to, :int64)};\n"
|
|
1409
2124
|
text += "#{inner}#{COMPUTATION_C_TYPES.fetch(type)} #{lanes.first} = #{name}" +
|
|
1410
2125
|
lanes.drop(1).map { |lane| ", #{lane} = #{identity}" }.join + ";\n"
|
|
2126
|
+
rounds = lanes.each_with_index.map { |lane, offset|
|
|
2127
|
+
term = lane_expression(assignment.expression, assignment.local, offset)
|
|
2128
|
+
"#{inner} {\n" \
|
|
2129
|
+
"#{inner} const int64_t #{index} = #{base} + #{offset};\n" \
|
|
2130
|
+
"#{inner} #{lane} = #{emit(term, type)};\n" \
|
|
2131
|
+
"#{inner} }\n"
|
|
2132
|
+
}.join
|
|
1411
2133
|
text += "#{inner}for (; #{base} + #{PARTIAL_ACCUMULATORS} <= #{limit}; " \
|
|
1412
2134
|
"#{base} += #{PARTIAL_ACCUMULATORS}) {\n"
|
|
1413
|
-
text += inner_loop_guard(inner)
|
|
1414
|
-
lanes.each_with_index do |lane, offset|
|
|
1415
|
-
term = lane_expression(assignment.expression, name, lane)
|
|
1416
|
-
text += "#{inner} {\n"
|
|
1417
|
-
text += "#{inner} const int64_t #{index} = #{base} + #{offset};\n"
|
|
1418
|
-
text += "#{inner} #{lane} = #{emit(term, type)};\n"
|
|
1419
|
-
text += "#{inner} }\n"
|
|
1420
|
-
end
|
|
2135
|
+
text += inner_loop_guard(inner) + rounds
|
|
1421
2136
|
text += "#{inner}}\n"
|
|
1422
2137
|
text += "#{inner}#{name} = #{combine_partials(lanes, operator)};\n"
|
|
2138
|
+
tail = emit_statement(assignment, inner + " ")
|
|
1423
2139
|
text += "#{inner}for (int64_t #{index} = #{base}; #{index} < #{limit}; " \
|
|
1424
2140
|
"#{index}++) {\n"
|
|
1425
|
-
text += inner_loop_guard(inner)
|
|
1426
|
-
text += emit_statement(assignment, inner + " ")
|
|
2141
|
+
text += inner_loop_guard(inner) + tail
|
|
1427
2142
|
text += "#{inner}}\n"
|
|
1428
2143
|
text + "#{indent}}\n"
|
|
1429
2144
|
end
|
|
@@ -1431,11 +2146,12 @@ class CArray
|
|
|
1431
2146
|
# The same term, folding into one chain instead of into the
|
|
1432
2147
|
# accumulator. The accumulator appears once and at the top, which is
|
|
1433
2148
|
# what reduction_accumulation checked, so only that operand moves.
|
|
1434
|
-
def lane_expression (expression,
|
|
2149
|
+
def lane_expression (expression, local, lane)
|
|
1435
2150
|
operands = [expression.left, expression.right].map { |operand|
|
|
1436
|
-
next operand unless accumulator?(operand,
|
|
2151
|
+
next operand unless accumulator?(operand, local)
|
|
1437
2152
|
read = LocalRead.new(operand.name, operand.location)
|
|
1438
|
-
read.
|
|
2153
|
+
read.binding = operand.binding
|
|
2154
|
+
read.lane = lane
|
|
1439
2155
|
read.type = operand.type
|
|
1440
2156
|
read
|
|
1441
2157
|
}
|
|
@@ -1481,26 +2197,28 @@ class CArray
|
|
|
1481
2197
|
end
|
|
1482
2198
|
|
|
1483
2199
|
def emit_assignment (assignment, indent)
|
|
1484
|
-
name = assignment.
|
|
2200
|
+
name = local_c_name(assignment.name, assignment.binding)
|
|
1485
2201
|
type = assignment.type
|
|
1486
|
-
declared = @declared_locals.key?(name)
|
|
1487
|
-
@declared_locals[name] = type
|
|
1488
2202
|
|
|
1489
2203
|
# A local carries a mask alongside its value, so that reading it later
|
|
1490
|
-
# is the same as reading what it was computed from
|
|
2204
|
+
# is the same as reading what it was computed from -- and, as a cell
|
|
2205
|
+
# written here would, the masks of the branches this assignment stands
|
|
2206
|
+
# in. An `if` or a `while` taken on a missing cell was decided by
|
|
2207
|
+
# bytes that mean nothing, so what it writes means nothing either,
|
|
2208
|
+
# wherever it writes it. Without this the rule held for
|
|
2209
|
+
# `out[i] = 1.0` and not for `v = 1.0; out[i] = v`, which is the
|
|
2210
|
+
# spelling a search over a row takes.
|
|
1491
2211
|
mask = if @masked
|
|
1492
|
-
"#{indent}#{
|
|
1493
|
-
"#{
|
|
2212
|
+
"#{indent}#{local_mask_name(name)} = " \
|
|
2213
|
+
"#{combine_masks(@carried_masks + [emit_mask(assignment.expression)])};\n"
|
|
1494
2214
|
else
|
|
1495
2215
|
""
|
|
1496
2216
|
end
|
|
1497
2217
|
|
|
1498
2218
|
if assignment.expression.is_a?(Conditional)
|
|
1499
|
-
|
|
1500
|
-
lines + emit_conditional_statement(assignment.expression, name, type, indent) + mask
|
|
2219
|
+
emit_conditional_statement(assignment.expression, name, type, indent) + mask
|
|
1501
2220
|
else
|
|
1502
|
-
|
|
1503
|
-
"#{indent}#{prefix}#{name} = #{emit(assignment.expression, type)};\n" + mask
|
|
2221
|
+
"#{indent}#{name} = #{emit(assignment.expression, type)};\n" + mask
|
|
1504
2222
|
end
|
|
1505
2223
|
end
|
|
1506
2224
|
|
|
@@ -1519,6 +2237,213 @@ class CArray
|
|
|
1519
2237
|
subscripts || @analyzer.index_names.map { |name| [name, 0] }
|
|
1520
2238
|
end
|
|
1521
2239
|
|
|
2240
|
+
# `sum(w)` / `min(w)` / `max(w)`: a call to a helper in the preamble.
|
|
2241
|
+
#
|
|
2242
|
+
# A call rather than a loop unrolled at the statement, because these
|
|
2243
|
+
# stand in expressions -- `max(w) - min(w)` is one line with two of them
|
|
2244
|
+
# -- and an expression has nowhere to put a loop. What it costs is
|
|
2245
|
+
# nothing: the helper is `static inline` and the length is a literal, so
|
|
2246
|
+
# the compiler has the same code in front of it either way.
|
|
2247
|
+
def intrinsic_reference (node)
|
|
2248
|
+
need(node.intrinsic, node.storage, nil)
|
|
2249
|
+
"#{intrinsic_helper_name(node.intrinsic, node.storage)}" \
|
|
2250
|
+
"(#{local_c_name(node.name, node.binding)}, " \
|
|
2251
|
+
"#{local_array_extent_text(node, 0)})"
|
|
2252
|
+
end
|
|
2253
|
+
|
|
2254
|
+
# A length the kernel works out cannot choose a network -- the choice
|
|
2255
|
+
# is made as the C is written and a network is a fixed run of
|
|
2256
|
+
# comparators -- so it sorts by the helper that takes its length as an
|
|
2257
|
+
# argument, which is the insertion sort.
|
|
2258
|
+
def emit_intrinsic_statement (statement, indent)
|
|
2259
|
+
cells = statement.shape.first
|
|
2260
|
+
name = local_c_name(statement.name, statement.binding)
|
|
2261
|
+
# One cell is in order already, and a network of no comparators is
|
|
2262
|
+
# nothing to call.
|
|
2263
|
+
return "" if cells == 1
|
|
2264
|
+
unless cells.is_a?(Integer)
|
|
2265
|
+
need(:sort, statement.storage, nil)
|
|
2266
|
+
return "#{indent}#{intrinsic_helper_name(:sort, statement.storage)}" \
|
|
2267
|
+
"(#{name}, #{local_array_extent_text(statement, 0)});\n"
|
|
2268
|
+
end
|
|
2269
|
+
if SortingNetworks.covers?(cells)
|
|
2270
|
+
need(:sort, statement.storage, cells)
|
|
2271
|
+
"#{indent}#{intrinsic_helper_name(:sort, statement.storage)}" \
|
|
2272
|
+
"_#{cells}(#{name});\n"
|
|
2273
|
+
else
|
|
2274
|
+
need(:sort, statement.storage, nil)
|
|
2275
|
+
"#{indent}#{intrinsic_helper_name(:sort, statement.storage)}" \
|
|
2276
|
+
"(#{name}, #{cells});\n"
|
|
2277
|
+
end
|
|
2278
|
+
end
|
|
2279
|
+
|
|
2280
|
+
def intrinsic_helper_name (intrinsic, storage)
|
|
2281
|
+
"carray_jit_#{intrinsic}_#{storage}"
|
|
2282
|
+
end
|
|
2283
|
+
|
|
2284
|
+
# What the preamble has to write out. A list used as a set, the way
|
|
2285
|
+
# `@clamp_types` and `@gamma_types` are, so that two sorts of one type
|
|
2286
|
+
# and length are one helper.
|
|
2287
|
+
def need (intrinsic, storage, cells)
|
|
2288
|
+
entry = [intrinsic, storage, cells]
|
|
2289
|
+
@intrinsic_needs << entry unless @intrinsic_needs.include?(entry)
|
|
2290
|
+
end
|
|
2291
|
+
|
|
2292
|
+
# `w = CArray.double(9)` where the statement stands. The declaration is
|
|
2293
|
+
# at the head of the block; what is left here is what Ruby does at this
|
|
2294
|
+
# line, which for the zeroed spellings is a fresh array of zeros.
|
|
2295
|
+
#
|
|
2296
|
+
# `sizeof` rather than a byte count written out: the length is already
|
|
2297
|
+
# in the declaration, and one place for it is one place to be wrong.
|
|
2298
|
+
# The shadow is cleared with the cells, and a cleared shadow byte is
|
|
2299
|
+
# "not missing": a fresh array in Ruby holds zeros, which are values
|
|
2300
|
+
# that are there. The `empty` spelling says nothing about either, so
|
|
2301
|
+
# its shadow is whatever the stack held -- the same bargain its cells
|
|
2302
|
+
# are under, and the same promise the block made by writing `empty`.
|
|
2303
|
+
def emit_local_array_clearing (declaration, indent)
|
|
2304
|
+
return "" unless declaration.zeroed
|
|
2305
|
+
name = local_c_name(declaration.name, declaration.binding)
|
|
2306
|
+
# `sizeof` answers for an array and not for a pointer, so one the
|
|
2307
|
+
# kernel allocated is cleared by the count it was allocated with.
|
|
2308
|
+
if declaration.heap
|
|
2309
|
+
cells = local_array_cells_text(name, declaration.shape)
|
|
2310
|
+
bytes = "sizeof(#{storage_c_type(declaration.storage)}) * " \
|
|
2311
|
+
"(size_t) #{cells}"
|
|
2312
|
+
return "#{indent}memset(#{name}, 0, #{bytes});\n" +
|
|
2313
|
+
(@masked ?
|
|
2314
|
+
"#{indent}memset(#{local_mask_name(name)}, 0, " \
|
|
2315
|
+
"(size_t) #{cells});\n" : "")
|
|
2316
|
+
end
|
|
2317
|
+
"#{indent}memset(#{name}, 0, sizeof #{name});\n" +
|
|
2318
|
+
(@masked ?
|
|
2319
|
+
"#{indent}memset(#{local_mask_name(name)}, 0, " \
|
|
2320
|
+
"sizeof #{local_mask_name(name)});\n" : "")
|
|
2321
|
+
end
|
|
2322
|
+
|
|
2323
|
+
# `w[k] = UNDEF`: the shadow byte alone. The cell's bytes are out of
|
|
2324
|
+
# contract once it is marked, so there is nothing to store into them.
|
|
2325
|
+
def emit_local_array_mask_write (write, indent)
|
|
2326
|
+
"#{indent}#{local_array_reference(write, :mask)} = 1;\n"
|
|
2327
|
+
end
|
|
2328
|
+
|
|
2329
|
+
def emit_local_array_write (write, indent)
|
|
2330
|
+
target = local_array_reference(write)
|
|
2331
|
+
expression = write.expression
|
|
2332
|
+
# A cell of a local array carries a mask beside it the way a scalar
|
|
2333
|
+
# local carries one beside its value, and by the same rule: what the
|
|
2334
|
+
# expression written into it carried, and the branches the write
|
|
2335
|
+
# stands in. A write reached because a missing cell said so was
|
|
2336
|
+
# reached for no reason, whichever of the three it writes.
|
|
2337
|
+
mask = if @masked
|
|
2338
|
+
"#{indent}#{local_array_reference(write, :mask)} = " \
|
|
2339
|
+
"#{combine_masks(@carried_masks + [emit_mask(write.expression)])};\n"
|
|
2340
|
+
else
|
|
2341
|
+
""
|
|
2342
|
+
end
|
|
2343
|
+
if expression.is_a?(Conditional)
|
|
2344
|
+
temporary = next_temporary
|
|
2345
|
+
type = expression.type
|
|
2346
|
+
"#{indent}#{COMPUTATION_C_TYPES.fetch(type)} #{temporary};\n" +
|
|
2347
|
+
emit_conditional_statement(expression, temporary, type, indent) +
|
|
2348
|
+
"#{indent}#{target} = " \
|
|
2349
|
+
"#{cast_to_storage(temporary, type, write.storage)};\n" + mask
|
|
2350
|
+
else
|
|
2351
|
+
value = cast_to_storage(emit(expression, expression.type),
|
|
2352
|
+
expression.type, write.storage)
|
|
2353
|
+
"#{indent}#{target} = #{value};\n" + mask
|
|
2354
|
+
end
|
|
2355
|
+
end
|
|
2356
|
+
|
|
2357
|
+
# `w[(k) + 1]`, and for a rank above one the row-major offset with the
|
|
2358
|
+
# strides folded in -- the shape is written in the block, so they are
|
|
2359
|
+
# constants rather than numbers the kernel is handed.
|
|
2360
|
+
# `:mask` asks for the shadow byte beside the cell rather than the
|
|
2361
|
+
# cell: the same position, in the array beside it. It reports no index
|
|
2362
|
+
# of its own, for the reason a captured array's mask read does not --
|
|
2363
|
+
# the value's read asks the same question a moment later, and one
|
|
2364
|
+
# report is the one the cell deserves.
|
|
2365
|
+
def local_array_reference (node, kind = :data)
|
|
2366
|
+
name = local_c_name(node.name, node.binding)
|
|
2367
|
+
name = local_mask_name(name) if kind == :mask
|
|
2368
|
+
terms = node.subscripts.each_with_index.map { |(index, offset), axis|
|
|
2369
|
+
position = local_array_position(node, index, offset, axis,
|
|
2370
|
+
reporting: kind != :mask)
|
|
2371
|
+
stride = local_array_stride_text(node, axis)
|
|
2372
|
+
stride == "1" ? position : "(#{position}) * #{stride}"
|
|
2373
|
+
}
|
|
2374
|
+
"#{name}[#{terms.join(' + ')}]"
|
|
2375
|
+
end
|
|
2376
|
+
|
|
2377
|
+
# How long an axis of a local array is, as the C says it: the number
|
|
2378
|
+
# the block wrote, or the name the kernel worked the length out into.
|
|
2379
|
+
def local_array_extent_text (node, axis)
|
|
2380
|
+
extent = node.shape.fetch(axis)
|
|
2381
|
+
return extent.to_s if extent.is_a?(Integer)
|
|
2382
|
+
heap_extent_name(local_c_name(node.name, node.binding), axis)
|
|
2383
|
+
end
|
|
2384
|
+
|
|
2385
|
+
# And how many cells it has altogether, for the allocation and the
|
|
2386
|
+
# clearing.
|
|
2387
|
+
def local_array_cells_text (variable, shape)
|
|
2388
|
+
return shape.inject(1, :*).to_s if literal_shape?(shape)
|
|
2389
|
+
heap_cells_name(variable)
|
|
2390
|
+
end
|
|
2391
|
+
|
|
2392
|
+
# How far one step along an axis moves: the cells of every axis inside
|
|
2393
|
+
# it, which are numbers where the block wrote numbers and the lengths
|
|
2394
|
+
# the kernel worked out where it did not.
|
|
2395
|
+
def local_array_stride_text (node, axis)
|
|
2396
|
+
inner = node.shape[(axis + 1)..]
|
|
2397
|
+
return inner.inject(1, :*).to_s if literal_shape?(inner)
|
|
2398
|
+
inner.each_with_index.map { |extent, offset|
|
|
2399
|
+
extent.is_a?(Integer) ? extent.to_s :
|
|
2400
|
+
heap_extent_name(local_c_name(node.name, node.binding),
|
|
2401
|
+
axis + 1 + offset)
|
|
2402
|
+
}.join(" * ")
|
|
2403
|
+
end
|
|
2404
|
+
|
|
2405
|
+
# An axis the analyzer settled carries an index and a literal offset,
|
|
2406
|
+
# and the C says so with nothing around it. One it could not settle
|
|
2407
|
+
# carries the expression, and the check is here.
|
|
2408
|
+
def local_array_position (node, index, offset, axis, reporting: true)
|
|
2409
|
+
settled = node.shape.fetch(axis).is_a?(Integer)
|
|
2410
|
+
# The write around this has already worked the position out and
|
|
2411
|
+
# tested it.
|
|
2412
|
+
guarded = @guarded_axes[node]
|
|
2413
|
+
return guarded.fetch(axis) if guarded&.key?(axis)
|
|
2414
|
+
if index.nil?
|
|
2415
|
+
return offset.to_s unless offset.is_a?(Node)
|
|
2416
|
+
# A position that is a number is settled against a length that is
|
|
2417
|
+
# one. Against a length the kernel works out it is not: 3 is
|
|
2418
|
+
# inside an array of 4 cells and outside one of 2, so it is checked
|
|
2419
|
+
# where the cell is reached like any other.
|
|
2420
|
+
return emit(offset, :int64) if settled &&
|
|
2421
|
+
Analyzer.fixed_subscript?(offset)
|
|
2422
|
+
# A write has already worked this position out and tested it.
|
|
2423
|
+
held = @position_temporaries[offset]
|
|
2424
|
+
return held if held
|
|
2425
|
+
@uses_index_check = true
|
|
2426
|
+
return "carray_jit_index(#{emit(offset, :int64)}, " \
|
|
2427
|
+
"#{local_array_extent_text(node, axis)}, " \
|
|
2428
|
+
"#{reporting ? error_argument : '(int32_t *) 0'})"
|
|
2429
|
+
end
|
|
2430
|
+
# An index on an axis whose length the kernel works out was not
|
|
2431
|
+
# settled either: (a) had no number to compare the loop's reach
|
|
2432
|
+
# against, so the reach is tested here.
|
|
2433
|
+
unless settled
|
|
2434
|
+
@uses_index_check = true
|
|
2435
|
+
position = if offset.zero? then index.to_s
|
|
2436
|
+
elsif offset.negative? then "#{index} - #{-offset}"
|
|
2437
|
+
else "#{index} + #{offset}"
|
|
2438
|
+
end
|
|
2439
|
+
return "carray_jit_index(#{position}, " \
|
|
2440
|
+
"#{local_array_extent_text(node, axis)}, " \
|
|
2441
|
+
"#{reporting ? error_argument : '(int32_t *) 0'})"
|
|
2442
|
+
end
|
|
2443
|
+
return index.to_s if offset.zero?
|
|
2444
|
+
offset.negative? ? "#{index} - #{-offset}" : "#{index} + #{offset}"
|
|
2445
|
+
end
|
|
2446
|
+
|
|
1522
2447
|
def emit_element_write (write, indent)
|
|
1523
2448
|
unless @masked
|
|
1524
2449
|
@masked_flag = nil
|
|
@@ -1547,13 +2472,15 @@ class CArray
|
|
|
1547
2472
|
when ElementRead
|
|
1548
2473
|
cell_reference(node.array, node.subscripts, :mask)
|
|
1549
2474
|
when LocalRead
|
|
1550
|
-
local_mask_name(node.
|
|
2475
|
+
local_mask_name(local_c_name(node.name, node.binding, node.lane))
|
|
2476
|
+
when LocalArrayRead
|
|
2477
|
+
local_array_reference(node, :mask)
|
|
1551
2478
|
when Conditional
|
|
1552
2479
|
branches = "#{emit(node.condition, :boolean)} ? " \
|
|
1553
2480
|
"#{emit_mask(node.consequent)} : #{emit_mask(node.alternative)}"
|
|
1554
2481
|
combine_masks([emit_mask(node.condition), "(#{branches})"])
|
|
1555
2482
|
when IntegerLiteral, FloatLiteral, IndexVariable, CaptureRead, MaskTest,
|
|
1556
|
-
BoundsValue, ZeroLike
|
|
2483
|
+
LocalArrayMaskTest, BoundsValue, ZeroLike
|
|
1557
2484
|
"0"
|
|
1558
2485
|
else
|
|
1559
2486
|
combine_masks(node.children.map { |child| emit_mask(child) })
|
|
@@ -1578,14 +2505,24 @@ class CArray
|
|
|
1578
2505
|
type = expression.type
|
|
1579
2506
|
"#{indent}#{COMPUTATION_C_TYPES.fetch(type)} #{temporary};\n" +
|
|
1580
2507
|
emit_conditional_statement(expression, temporary, type, indent) +
|
|
1581
|
-
"#{indent}#{target} =
|
|
2508
|
+
"#{indent}#{target} = " \
|
|
2509
|
+
"#{cast_to_storage(temporary, type, array_storage(write.array))};\n"
|
|
1582
2510
|
else
|
|
2511
|
+
storage = array_storage(write.array)
|
|
2512
|
+
real = REAL_STORAGE_TYPES[storage]
|
|
2513
|
+
if real && rounded_real?(expression, real)
|
|
2514
|
+
return "#{indent}#{target} = #{emit(expression, real)};\n"
|
|
2515
|
+
end
|
|
1583
2516
|
value = cast_to_storage(emit(expression, expression.type),
|
|
1584
|
-
expression.type,
|
|
2517
|
+
expression.type, storage)
|
|
1585
2518
|
"#{indent}#{target} = #{value};\n"
|
|
1586
2519
|
end
|
|
1587
2520
|
end
|
|
1588
2521
|
|
|
2522
|
+
# A float cell, by what it computes in: the storage a rounding written
|
|
2523
|
+
# straight into keeps as a real number (see #rounded_real?).
|
|
2524
|
+
REAL_STORAGE_TYPES = { "float64" => :double, "float32" => :float }.freeze
|
|
2525
|
+
|
|
1589
2526
|
# A conditional in statement position becomes if/else rather than a
|
|
1590
2527
|
# ternary; it reads far better when the branches are long.
|
|
1591
2528
|
def emit_conditional_statement (node, target, type, indent)
|
|
@@ -1596,31 +2533,230 @@ class CArray
|
|
|
1596
2533
|
"#{indent}}\n"
|
|
1597
2534
|
end
|
|
1598
2535
|
|
|
1599
|
-
|
|
1600
|
-
|
|
2536
|
+
# What a cell is held in, and the casts between that and the type the
|
|
2537
|
+
# kernel computes in.
|
|
2538
|
+
#
|
|
2539
|
+
# Each is asked about a storage type rather than about an array's name.
|
|
2540
|
+
# An operand's storage type is looked up under its name, but not every
|
|
2541
|
+
# array a body reaches is an operand -- so a helper that did the lookup
|
|
2542
|
+
# itself could only answer for the ones that are, and the answer for
|
|
2543
|
+
# anything else would be a KeyError out of the middle of code
|
|
2544
|
+
# generation. The name that has one is looked up by #array_storage at
|
|
2545
|
+
# the call site.
|
|
2546
|
+
def storage_c_type (storage)
|
|
2547
|
+
STORAGE_C_TYPES.fetch(storage)
|
|
2548
|
+
end
|
|
2549
|
+
|
|
2550
|
+
# ---- the intrinsics, written into the preamble ----
|
|
2551
|
+
|
|
2552
|
+
# The accumulator `min` and `max` start from over an integer array,
|
|
2553
|
+
# where starting at the limit of the type rather than at the first cell
|
|
2554
|
+
# saves asking whether a cell has been seen yet. A floating array
|
|
2555
|
+
# starts at NaN instead, which any number displaces -- see
|
|
2556
|
+
# #intrinsic_fold_helper.
|
|
2557
|
+
INTRINSIC_LIMITS = {
|
|
2558
|
+
:int64 => ["INT64_MAX", "INT64_MIN"],
|
|
2559
|
+
:uint64 => ["UINT64_MAX", "0"],
|
|
2560
|
+
}.freeze
|
|
2561
|
+
|
|
2562
|
+
def computation_c_type_of (storage)
|
|
2563
|
+
COMPUTATION_C_TYPES.fetch(TypeAssignment.storage_type(storage))
|
|
2564
|
+
end
|
|
2565
|
+
|
|
2566
|
+
def floating_storage? (storage)
|
|
2567
|
+
type = TypeAssignment.storage_type(storage)
|
|
2568
|
+
[:float, :double].include?(type)
|
|
2569
|
+
end
|
|
2570
|
+
|
|
2571
|
+
def intrinsic_helpers
|
|
2572
|
+
return "" if @intrinsic_needs.empty?
|
|
2573
|
+
text = +""
|
|
2574
|
+
%i[sum min max].each do |intrinsic|
|
|
2575
|
+
@intrinsic_needs.select { |kind, _, _| kind == intrinsic }
|
|
2576
|
+
.map { |_, storage, _| storage }.uniq.each do |storage|
|
|
2577
|
+
text << intrinsic_fold_helper(intrinsic, storage)
|
|
2578
|
+
end
|
|
2579
|
+
end
|
|
2580
|
+
sorted = @intrinsic_needs.select { |kind, _, _| kind == :sort }
|
|
2581
|
+
sorted.map { |_, storage, _| storage }.uniq.each do |storage|
|
|
2582
|
+
text << intrinsic_order_helper(storage)
|
|
2583
|
+
end
|
|
2584
|
+
sorted.each do |_, storage, cells|
|
|
2585
|
+
text << (cells ? intrinsic_network_helper(storage, cells)
|
|
2586
|
+
: intrinsic_insertion_helper(storage))
|
|
2587
|
+
end
|
|
2588
|
+
text
|
|
2589
|
+
end
|
|
2590
|
+
|
|
2591
|
+
def intrinsic_fold_helper (intrinsic, storage)
|
|
2592
|
+
cell = storage_c_type(storage)
|
|
2593
|
+
result = computation_c_type_of(storage)
|
|
2594
|
+
type = TypeAssignment.storage_type(storage)
|
|
2595
|
+
if intrinsic == :sum
|
|
2596
|
+
return <<~C
|
|
2597
|
+
/* `sum(w)`, added in index order from the first cell: what the
|
|
2598
|
+
same loop written out in Ruby adds, in the same order, so the
|
|
2599
|
+
last bit is the same one. No partial sums -- a local array is
|
|
2600
|
+
small enough that splitting buys nothing and would change the
|
|
2601
|
+
answer. */
|
|
2602
|
+
static inline #{result}
|
|
2603
|
+
carray_jit_sum_#{storage} (const #{cell} *w, int64_t n)
|
|
2604
|
+
{
|
|
2605
|
+
#{result} total = 0;
|
|
2606
|
+
for ( int64_t k = 0; k < n; k++ ) total += w[k];
|
|
2607
|
+
return total;
|
|
2608
|
+
}
|
|
2609
|
+
|
|
2610
|
+
C
|
|
2611
|
+
end
|
|
2612
|
+
if floating_storage?(storage)
|
|
2613
|
+
comparison = intrinsic == :min ? "<" : ">"
|
|
2614
|
+
return <<~C
|
|
2615
|
+
/* `#{intrinsic}(w)`, which is what CArray's own #{intrinsic} answers.
|
|
2616
|
+
A cell takes the accumulator only by beating it outright, or by
|
|
2617
|
+
being a number where the accumulator is still NaN: so a NaN is
|
|
2618
|
+
skipped wherever it stands, an array of nothing but NaN comes
|
|
2619
|
+
back NaN, and of two cells that compare equal the first is kept
|
|
2620
|
+
-- which is how CArray keeps 0.0 or -0.0, whichever came first.
|
|
2621
|
+
fmin and fmax answer the same but for that: C leaves which zero
|
|
2622
|
+
they give up to the library. Starting at NaN is what lets the
|
|
2623
|
+
first cell in take the accumulator without a seen-yet test. */
|
|
2624
|
+
static inline #{result}
|
|
2625
|
+
carray_jit_#{intrinsic}_#{storage} (const #{cell} *w, int64_t n)
|
|
2626
|
+
{
|
|
2627
|
+
#{result} best = NAN;
|
|
2628
|
+
for ( int64_t k = 0; k < n; k++ ) {
|
|
2629
|
+
if ( w[k] #{comparison} best || best != best ) best = w[k];
|
|
2630
|
+
}
|
|
2631
|
+
return best;
|
|
2632
|
+
}
|
|
2633
|
+
|
|
2634
|
+
C
|
|
2635
|
+
end
|
|
2636
|
+
limit = INTRINSIC_LIMITS.fetch(type)[intrinsic == :min ? 0 : 1]
|
|
2637
|
+
comparison = intrinsic == :min ? "<" : ">"
|
|
2638
|
+
<<~C
|
|
2639
|
+
/* `#{intrinsic}(w)`, which is what CArray's own #{intrinsic} answers.
|
|
2640
|
+
An integer type has no NaN, so the limit this starts from is only
|
|
2641
|
+
ever an answer for an array with no cells. */
|
|
2642
|
+
static inline #{result}
|
|
2643
|
+
carray_jit_#{intrinsic}_#{storage} (const #{cell} *w, int64_t n)
|
|
2644
|
+
{
|
|
2645
|
+
#{result} best = #{limit};
|
|
2646
|
+
for ( int64_t k = 0; k < n; k++ ) best = (w[k] #{comparison} best) ? w[k] : best;
|
|
2647
|
+
return best;
|
|
2648
|
+
}
|
|
2649
|
+
|
|
2650
|
+
C
|
|
2651
|
+
end
|
|
2652
|
+
|
|
2653
|
+
# The order the two sorts put the cells in, in one place so that the
|
|
2654
|
+
# network and the loop cannot disagree about it.
|
|
2655
|
+
#
|
|
2656
|
+
# A NaN comes after every number and ties with another NaN, which makes
|
|
2657
|
+
# this a total preorder -- and a comparator network sorts under any of
|
|
2658
|
+
# those, so the finite cells come out ascending with the NaNs behind
|
|
2659
|
+
# them. That is where CArray's own sort puts them.
|
|
2660
|
+
#
|
|
2661
|
+
# The min / max folds above skip a NaN, which drops a cell rather than
|
|
2662
|
+
# moving it. A fold wants that; a sort has to put every cell somewhere.
|
|
2663
|
+
def intrinsic_order_helper (storage)
|
|
2664
|
+
cell = storage_c_type(storage)
|
|
2665
|
+
nan_term = floating_storage?(storage) ? " || (a != a && b == b)" : ""
|
|
2666
|
+
<<~C
|
|
2667
|
+
/* Whether `a` belongs after `b` once the cells are in order.#{
|
|
2668
|
+
floating_storage?(storage) ?
|
|
2669
|
+
"\n A NaN belongs after every number, and two NaNs tie." : ""} */
|
|
2670
|
+
static inline int
|
|
2671
|
+
carray_jit_after_#{storage} (#{cell} a, #{cell} b)
|
|
2672
|
+
{
|
|
2673
|
+
return (b < a)#{nan_term};
|
|
2674
|
+
}
|
|
2675
|
+
|
|
2676
|
+
/* One compare-exchange: the smaller of the two ends up in `low`.
|
|
2677
|
+
Written as two selects rather than a branch, so the sequence a
|
|
2678
|
+
network lays out has nothing in it to predict. */
|
|
2679
|
+
static inline void
|
|
2680
|
+
carray_jit_cx_#{storage} (#{cell} *low, #{cell} *high)
|
|
2681
|
+
{
|
|
2682
|
+
#{cell} a = *low, b = *high;
|
|
2683
|
+
int swap = carray_jit_after_#{storage}(a, b);
|
|
2684
|
+
*low = swap ? b : a;
|
|
2685
|
+
*high = swap ? a : b;
|
|
2686
|
+
}
|
|
2687
|
+
|
|
2688
|
+
C
|
|
2689
|
+
end
|
|
2690
|
+
|
|
2691
|
+
def intrinsic_network_helper (storage, cells)
|
|
2692
|
+
network = SortingNetworks.for(cells)
|
|
2693
|
+
pairs = network.map { |low, high|
|
|
2694
|
+
" carray_jit_cx_#{storage}(&w[#{low}], &w[#{high}]);"
|
|
2695
|
+
}.join("\n")
|
|
2696
|
+
<<~C
|
|
2697
|
+
/* `sort(w)` for #{cells} cells: a comparator network, which is the
|
|
2698
|
+
same #{network.size} compare-exchanges every time in the same
|
|
2699
|
+
order -- no branch, nothing to predict, and the smallest number
|
|
2700
|
+
of comparisons known for this length. */
|
|
2701
|
+
static inline void
|
|
2702
|
+
carray_jit_sort_#{storage}_#{cells} (#{storage_c_type(storage)} *w)
|
|
2703
|
+
{
|
|
2704
|
+
#{pairs}
|
|
2705
|
+
}
|
|
2706
|
+
|
|
2707
|
+
C
|
|
2708
|
+
end
|
|
2709
|
+
|
|
2710
|
+
def intrinsic_insertion_helper (storage)
|
|
2711
|
+
cell = storage_c_type(storage)
|
|
2712
|
+
<<~C
|
|
2713
|
+
/* `sort(w)` past the length the networks cover: an insertion sort,
|
|
2714
|
+
which is a loop. The tables would keep growing and what a
|
|
2715
|
+
network buys shrinks -- long before this there are more
|
|
2716
|
+
comparators than the machine has registers. */
|
|
2717
|
+
static inline void
|
|
2718
|
+
carray_jit_sort_#{storage} (#{cell} *w, int64_t n)
|
|
2719
|
+
{
|
|
2720
|
+
for ( int64_t k = 1; k < n; k++ ) {
|
|
2721
|
+
#{cell} value = w[k];
|
|
2722
|
+
int64_t j = k - 1;
|
|
2723
|
+
while ( j >= 0 && carray_jit_after_#{storage}(w[j], value) ) {
|
|
2724
|
+
w[j + 1] = w[j];
|
|
2725
|
+
j--;
|
|
2726
|
+
}
|
|
2727
|
+
w[j + 1] = value;
|
|
2728
|
+
}
|
|
2729
|
+
}
|
|
2730
|
+
|
|
2731
|
+
C
|
|
1601
2732
|
end
|
|
1602
2733
|
|
|
1603
|
-
|
|
2734
|
+
# The storage type of an operand, by the name the block reached it by.
|
|
2735
|
+
def array_storage (array)
|
|
2736
|
+
@storage_types.fetch(array)
|
|
2737
|
+
end
|
|
2738
|
+
|
|
2739
|
+
def cast_to_storage (text, type, storage)
|
|
1604
2740
|
# Storing into a boolean array normalises: CArray's boolean is a byte
|
|
1605
2741
|
# holding 0 or 1, and a kernel is not the place that invariant stops
|
|
1606
2742
|
# being true.
|
|
1607
|
-
return "(uint8_t)((#{text}) ? 1 : 0)" if boolean_storage?(
|
|
1608
|
-
target = storage_c_type(
|
|
2743
|
+
return "(uint8_t)((#{text}) ? 1 : 0)" if boolean_storage?(storage)
|
|
2744
|
+
target = storage_c_type(storage)
|
|
1609
2745
|
return text if COMPUTATION_C_TYPES.fetch(type) == target
|
|
1610
2746
|
"(#{target})(#{text})"
|
|
1611
2747
|
end
|
|
1612
2748
|
|
|
1613
2749
|
# The cast that takes a stored cell to the type the kernel computes in,
|
|
1614
2750
|
# or nil where the two already agree.
|
|
1615
|
-
def widening_cast (
|
|
2751
|
+
def widening_cast (storage, type)
|
|
1616
2752
|
return nil if type == :boolean
|
|
1617
2753
|
wanted = COMPUTATION_C_TYPES.fetch(type)
|
|
1618
|
-
return nil if storage_c_type(
|
|
2754
|
+
return nil if storage_c_type(storage) == wanted
|
|
1619
2755
|
"(#{wanted})"
|
|
1620
2756
|
end
|
|
1621
2757
|
|
|
1622
|
-
def boolean_storage? (
|
|
1623
|
-
|
|
2758
|
+
def boolean_storage? (storage)
|
|
2759
|
+
storage == "boolean"
|
|
1624
2760
|
end
|
|
1625
2761
|
|
|
1626
2762
|
# A Ruby name is not always a C one: `Foo::TABLE` names one thing in
|
|
@@ -1630,6 +2766,143 @@ class CArray
|
|
|
1630
2766
|
name.to_s.gsub("::", "__")
|
|
1631
2767
|
end
|
|
1632
2768
|
|
|
2769
|
+
# The identifier each name is written as where it is written on its own,
|
|
2770
|
+
# as a variable of the kernel's body.
|
|
2771
|
+
#
|
|
2772
|
+
# `c_name` is not enough there. The kernel's own parameters are `data`,
|
|
2773
|
+
# `error`, `strides`, `bounds` -- ordinary words a block may perfectly
|
|
2774
|
+
# well have used for a capture -- and C keeps a list of its own;
|
|
2775
|
+
# `RESERVED_NAMES` is already both, and was already consulted for an
|
|
2776
|
+
# index. A capture was not consulted about, and a block closing over
|
|
2777
|
+
# something called `data` compiled to C that redeclared the kernel's own
|
|
2778
|
+
# parameter. A leading underscore is C's to reserve as well, which is
|
|
2779
|
+
# what the made-up names here start with. And a C name that starts
|
|
2780
|
+
# `carray_jit_` is the generator's, whoever wrote it in Ruby.
|
|
2781
|
+
#
|
|
2782
|
+
# An index is refused for this rather than moved, because its name is
|
|
2783
|
+
# written in the block and read back in messages. A capture's is not:
|
|
2784
|
+
# what it is called in the C is nobody's business, so it is moved out of
|
|
2785
|
+
# the way instead of the caller being asked to rename a variable in code
|
|
2786
|
+
# that has nothing to do with the kernel.
|
|
2787
|
+
#
|
|
2788
|
+
# Worked out for the whole kernel at once, because where a moved name
|
|
2789
|
+
# lands is not a question about that name alone: a block closing over
|
|
2790
|
+
# both `data` and `carray_jit_name_data` would otherwise have the first
|
|
2791
|
+
# move onto the second. So every name that stays is claimed first, and
|
|
2792
|
+
# a move that would collide keeps going.
|
|
2793
|
+
#
|
|
2794
|
+
# Only a name that had to move moves, so the C a reader sees is what it
|
|
2795
|
+
# was. A decorated name (`p_a`, `a_s0`, `m_a`) could not collide with a
|
|
2796
|
+
# parameter or a keyword and is left alone.
|
|
2797
|
+
def bare_names
|
|
2798
|
+
@bare_names ||= begin
|
|
2799
|
+
names = (@reals + @integers + @complexes + @unsigned_integers +
|
|
2800
|
+
@address_arrays + @address_functions.keys).uniq
|
|
2801
|
+
taken = names.map { |name| c_name(name) }
|
|
2802
|
+
names.each_with_object({}) do |name, found|
|
|
2803
|
+
text = c_name(name)
|
|
2804
|
+
if RESERVED_NAMES.include?(name.to_sym) || text.start_with?("_") ||
|
|
2805
|
+
text.start_with?("carray_jit_")
|
|
2806
|
+
# Not the bare `carray_jit_` prefix: `carray_jit_error` is the
|
|
2807
|
+
# error flag, and a capture called `error` would land on it.
|
|
2808
|
+
text = "carray_jit_name_#{text}"
|
|
2809
|
+
text = "#{text}_" while taken.include?(text)
|
|
2810
|
+
taken << text
|
|
2811
|
+
end
|
|
2812
|
+
found[name] = text
|
|
2813
|
+
end
|
|
2814
|
+
end
|
|
2815
|
+
end
|
|
2816
|
+
|
|
2817
|
+
def bare_name (name)
|
|
2818
|
+
bare_names.fetch(name) { c_name(name) }
|
|
2819
|
+
end
|
|
2820
|
+
|
|
2821
|
+
# The names this generator writes of its own, beside the kernel's
|
|
2822
|
+
# parameters and C's keywords: every decoration of every array the
|
|
2823
|
+
# kernel reaches, at every axis it has, and a borrowed function's type.
|
|
2824
|
+
# All of them, whether or not this kernel ends up writing each one, so
|
|
2825
|
+
# that the set is settled before anything is emitted.
|
|
2826
|
+
def decorated_names
|
|
2827
|
+
@decorated_names ||=
|
|
2828
|
+
@arrays.flat_map { |array|
|
|
2829
|
+
[pointer_name(array), mask_pointer_name(array)] +
|
|
2830
|
+
(0...array_rank(array)).flat_map { |axis|
|
|
2831
|
+
[stride_name(array, axis), mask_stride_name(array, axis),
|
|
2832
|
+
extent_name(array, axis)]
|
|
2833
|
+
}
|
|
2834
|
+
} + @address_functions.keys.map { |name| c_function_type_name(name) }
|
|
2835
|
+
end
|
|
2836
|
+
|
|
2837
|
+
# Whether a name is one the generator writes, or could. Everything it
|
|
2838
|
+
# makes up that is not a parameter, a keyword or a decoration has `__`
|
|
2839
|
+
# in it or starts `carray_jit_` -- the suffixes a local's bindings take,
|
|
2840
|
+
# an inner index's second name, the temporaries -- so those two are
|
|
2841
|
+
# taken whole.
|
|
2842
|
+
def generated_name? (text)
|
|
2843
|
+
RESERVED_NAMES.include?(text.to_sym) || text.include?("__") ||
|
|
2844
|
+
text.start_with?("carray_jit_") || decorated_names.include?(text)
|
|
2845
|
+
end
|
|
2846
|
+
|
|
2847
|
+
# The C name of each local the block assigns, by its Ruby name. A local
|
|
2848
|
+
# is ordinary Ruby and may be called anything, including a name the
|
|
2849
|
+
# generator writes -- `a_n0`, `error`, `int`, `x__2` -- and a local of
|
|
2850
|
+
# that name declared in the kernel's body hid what it named. So it is
|
|
2851
|
+
# moved, to `carray_jit_name<K>_<name>`, K counting the moved locals from
|
|
2852
|
+
# 1 in the order the body first assigns them, so the C -- and the cache
|
|
2853
|
+
# key it is part of -- is the same every time. One that does not
|
|
2854
|
+
# collide keeps the name the block gave it, which is what a reader of
|
|
2855
|
+
# the C looks for.
|
|
2856
|
+
#
|
|
2857
|
+
# The number is what keeps two moved locals apart once a suffix is on
|
|
2858
|
+
# them. A Ruby identifier cannot start with a digit, so the digits after
|
|
2859
|
+
# `carray_jit_name` run to the next `_`, and a different K is a different
|
|
2860
|
+
# C name whatever `__2`, `__p0` or `__mask` follows. A C name starting
|
|
2861
|
+
# `carray_jit_` is the generator's, so no capture is spelled like one.
|
|
2862
|
+
def local_c_names
|
|
2863
|
+
@local_c_names ||= begin
|
|
2864
|
+
moved = 0
|
|
2865
|
+
assigned_locals(@analyzer.body).uniq.to_h { |name|
|
|
2866
|
+
text = name.to_s
|
|
2867
|
+
text = "carray_jit_name#{moved += 1}_#{text}" if generated_name?(text)
|
|
2868
|
+
[name, text]
|
|
2869
|
+
}
|
|
2870
|
+
end
|
|
2871
|
+
end
|
|
2872
|
+
|
|
2873
|
+
# Every name the body binds, in the order it first binds them. An array
|
|
2874
|
+
# the body made is one of these: it is a local that happens to have
|
|
2875
|
+
# cells, and it is moved off the generator's own names for the same
|
|
2876
|
+
# reason and by the same numbering.
|
|
2877
|
+
def assigned_locals (node)
|
|
2878
|
+
own = case node
|
|
2879
|
+
when Assignment, LocalArrayDeclaration then [node.name]
|
|
2880
|
+
else []
|
|
2881
|
+
end
|
|
2882
|
+
own + node.children.compact.flat_map { |child| assigned_locals(child) }
|
|
2883
|
+
end
|
|
2884
|
+
|
|
2885
|
+
# A local's C variable is its name, moved or not, and then what says
|
|
2886
|
+
# which of its variables: `__2` for its second binding, `__p0` for a
|
|
2887
|
+
# partial accumulator of a split fold -- `carray_jit_name_error__2`.
|
|
2888
|
+
def local_c_name (name, binding, lane = nil)
|
|
2889
|
+
local_c_names.fetch(name) + (binding == 1 ? "" : "__#{binding}") +
|
|
2890
|
+
(lane ? "__p#{lane}" : "")
|
|
2891
|
+
end
|
|
2892
|
+
|
|
2893
|
+
# An index is refused where a local is moved: its name is the block's
|
|
2894
|
+
# text, and the messages speak of it. The parameters and C's keywords
|
|
2895
|
+
# were refused as the block was read; what is refused here needed the
|
|
2896
|
+
# arrays first.
|
|
2897
|
+
def refuse_generated_indices
|
|
2898
|
+
taken = @analyzer.index_names.select { |name| generated_name?(c_name(name)) }
|
|
2899
|
+
return if taken.empty?
|
|
2900
|
+
raise Unsupported.new(
|
|
2901
|
+
"#{taken.map { |name| "`#{name}`" }.join(', ')} " \
|
|
2902
|
+
"#{taken.size == 1 ? 'is a name' : 'are names'} the kernel's own C " \
|
|
2903
|
+
"uses, so #{taken.size == 1 ? 'it cannot be an index' : 'they cannot be indices'}")
|
|
2904
|
+
end
|
|
2905
|
+
|
|
1633
2906
|
def pointer_name (array)
|
|
1634
2907
|
"p_#{c_name(array)}"
|
|
1635
2908
|
end
|
|
@@ -1652,6 +2925,29 @@ class CArray
|
|
|
1652
2925
|
# what the kernel raises. Nothing in the C reaches Ruby to do it: the
|
|
1653
2926
|
# function still returns a number and touches no Ruby value, which is
|
|
1654
2927
|
# what lets its address be handed to a library, or called off the GVL.
|
|
2928
|
+
# The widths `clamp` is emitted for: the suffix its helper takes, the C
|
|
2929
|
+
# type, and whether that type has a NaN to refuse.
|
|
2930
|
+
CLAMP_C_TYPES = {
|
|
2931
|
+
:double => ["double", "double", true],
|
|
2932
|
+
:float => ["float", "float", true],
|
|
2933
|
+
:int64 => ["int64", "int64_t", false],
|
|
2934
|
+
:uint64 => ["uint64", "uint64_t", false],
|
|
2935
|
+
}.freeze
|
|
2936
|
+
|
|
2937
|
+
# The widths `Math.gamma` is emitted for: the helper's suffix, the C
|
|
2938
|
+
# type, and the C function it falls through to.
|
|
2939
|
+
GAMMA_C_TYPES = {
|
|
2940
|
+
:double => ["double", "double", "tgamma", "modf"],
|
|
2941
|
+
:float => ["float", "float", "tgammaf", "modff"],
|
|
2942
|
+
}.freeze
|
|
2943
|
+
|
|
2944
|
+
# Ruby answers `Math.gamma` for a small whole number from a table of
|
|
2945
|
+
# exact values rather than from tgamma, so the table is written into
|
|
2946
|
+
# the C -- filled from the Ruby that is compiling, the way `Math::PI`
|
|
2947
|
+
# is emitted as the double Ruby would have used. Its length is Ruby's
|
|
2948
|
+
# own: the table covers 1 through 23, and tgamma has the rest.
|
|
2949
|
+
GAMMA_TABLE_LIMIT = 23
|
|
2950
|
+
|
|
1655
2951
|
ERROR_FLAG = "carray_jit_error"
|
|
1656
2952
|
|
|
1657
2953
|
# A computation type this generator has no C for. Nothing reaches here
|
|
@@ -1666,18 +2962,19 @@ class CArray
|
|
|
1666
2962
|
end
|
|
1667
2963
|
|
|
1668
2964
|
def error_argument
|
|
2965
|
+
# Asking for the slot is what says this body can report, and every
|
|
2966
|
+
# place that can -- a checked subscript, the division helpers, a
|
|
2967
|
+
# `raise`, a pasted function that takes the flag -- asks here. So the
|
|
2968
|
+
# loops around it learn it from the statements they hold, which are
|
|
2969
|
+
# emitted before the loop's guard is written. A function's loops ask
|
|
2970
|
+
# too: they stop on its flag the way a kernel's stop on the slot.
|
|
2971
|
+
@body_reports = true
|
|
1669
2972
|
if @in_function
|
|
1670
2973
|
@uses_error_flag = true
|
|
1671
2974
|
# A pasted body reports into the kernel's slot, which reaches it as
|
|
1672
2975
|
# a parameter; a standalone one into the flag in its own object.
|
|
1673
2976
|
return @error_parameter ? ERROR_FLAG : "&#{ERROR_FLAG}"
|
|
1674
2977
|
end
|
|
1675
|
-
# Asking for the slot is what says this body can report, and every
|
|
1676
|
-
# place that can -- a checked subscript, the division helpers, a
|
|
1677
|
-
# `raise`, a pasted function that takes the flag -- asks here. So the
|
|
1678
|
-
# loops above learn it from the statements they will hold, which are
|
|
1679
|
-
# emitted before the loop is opened.
|
|
1680
|
-
@body_reports = true
|
|
1681
2978
|
@masked_flag ? "(#{@masked_flag} ? (int32_t *) 0 : error)" : "error"
|
|
1682
2979
|
end
|
|
1683
2980
|
|
|
@@ -1787,7 +3084,7 @@ class CArray
|
|
|
1787
3084
|
return "#{mask_pointer_name(array)}[#{terms.join(' + ')}]"
|
|
1788
3085
|
end
|
|
1789
3086
|
|
|
1790
|
-
type = storage_c_type(array)
|
|
3087
|
+
type = storage_c_type(array_storage(array))
|
|
1791
3088
|
outer = (0...(count - 1)).map { |axis|
|
|
1792
3089
|
index, offset = subscripts[axis]
|
|
1793
3090
|
"(#{index_expression(array, axis, index, offset)}) * " \
|
|
@@ -1810,11 +3107,17 @@ class CArray
|
|
|
1810
3107
|
# Emits `node` so that its value has type `target`, inserting a cast
|
|
1811
3108
|
# only where the type actually changes.
|
|
1812
3109
|
def emit (node, target)
|
|
3110
|
+
return emit_rounded_real(node, target).first if rounded_real?(node, target)
|
|
1813
3111
|
widen(*emit_raw(node), node.type, target).first
|
|
1814
3112
|
end
|
|
1815
3113
|
|
|
1816
3114
|
def emit_operand (node, target, parent_precedence, right_side = false)
|
|
1817
|
-
text, precedence =
|
|
3115
|
+
text, precedence =
|
|
3116
|
+
if rounded_real?(node, target)
|
|
3117
|
+
emit_rounded_real(node, target)
|
|
3118
|
+
else
|
|
3119
|
+
widen(*emit_raw(node), node.type, target)
|
|
3120
|
+
end
|
|
1818
3121
|
parenthesize(text, precedence, right_side ? parent_precedence + 1 : parent_precedence)
|
|
1819
3122
|
end
|
|
1820
3123
|
|
|
@@ -1822,13 +3125,32 @@ class CArray
|
|
|
1822
3125
|
# most of these on its own, but writing them down is what makes the
|
|
1823
3126
|
# dumped source say where the type changed -- and `int64_t` to
|
|
1824
3127
|
# `double _Complex` is a conversion worth seeing.
|
|
3128
|
+
#
|
|
3129
|
+
# `int64` and `uint64` are the exception the rank cannot express.
|
|
3130
|
+
# They are the same width and differ only in sign, so the rank puts
|
|
3131
|
+
# one above the other and a `uint64` reaching an `int64` context is a
|
|
3132
|
+
# step *down* it -- left uncast, on the reasoning that C would do it
|
|
3133
|
+
# anyway. C does not: in `i < n` with `i` an `int64_t` and `n` a
|
|
3134
|
+
# `size_t`, the conversion C performs is on the other operand, and a
|
|
3135
|
+
# loop counting down from -1 then runs against a bound of 2**64 - 1.
|
|
3136
|
+
# The compiler says so as `-Wsign-compare`, which is the warning every
|
|
3137
|
+
# generated loop over a `size_t` extent carried.
|
|
1825
3138
|
def widen (text, precedence, from, to)
|
|
1826
3139
|
here, there = NUMERIC_RANK[from], NUMERIC_RANK[to]
|
|
1827
|
-
return [text, precedence] if here.nil? || there.nil?
|
|
3140
|
+
return [text, precedence] if here.nil? || there.nil?
|
|
3141
|
+
unless there > here || crosses_sign?(from, to)
|
|
3142
|
+
return [text, precedence]
|
|
3143
|
+
end
|
|
1828
3144
|
["(#{COMPUTATION_C_TYPES.fetch(to)})" \
|
|
1829
3145
|
"#{parenthesize(text, precedence, LEAF_PRECEDENCE)}", UNARY_PRECEDENCE]
|
|
1830
3146
|
end
|
|
1831
3147
|
|
|
3148
|
+
def crosses_sign? (from, to)
|
|
3149
|
+
from != to &&
|
|
3150
|
+
TypeAssignment::INTEGER_TYPES.include?(from) &&
|
|
3151
|
+
TypeAssignment::INTEGER_TYPES.include?(to)
|
|
3152
|
+
end
|
|
3153
|
+
|
|
1832
3154
|
def parenthesize (text, precedence, needed)
|
|
1833
3155
|
precedence < needed ? "(#{text})" : text
|
|
1834
3156
|
end
|
|
@@ -1849,8 +3171,9 @@ class CArray
|
|
|
1849
3171
|
when BoundsValue then ["bounds[#{node.slot}]", LEAF_PRECEDENCE]
|
|
1850
3172
|
when ZeroLike
|
|
1851
3173
|
[ZEROES.fetch(node.type), LEAF_PRECEDENCE]
|
|
1852
|
-
when LocalRead then [node.
|
|
1853
|
-
|
|
3174
|
+
when LocalRead then [local_c_name(node.name, node.binding, node.lane),
|
|
3175
|
+
LEAF_PRECEDENCE]
|
|
3176
|
+
when CaptureRead then [bare_name(node.name), LEAF_PRECEDENCE]
|
|
1854
3177
|
# A read is widened to the type the kernel computes in, because that
|
|
1855
3178
|
# is the type the Ruby loop computes in: reading a float32 cell in
|
|
1856
3179
|
# Ruby gives a Float, and reading an int32 cell gives an Integer that
|
|
@@ -1861,9 +3184,20 @@ class CArray
|
|
|
1861
3184
|
# A boolean cell is a byte holding 0 or 1, so it is already the value
|
|
1862
3185
|
# it stands for. Keeping it that way on the way out is this kernel's
|
|
1863
3186
|
# business, and cast_to_storage does it.
|
|
3187
|
+
when IntrinsicCall
|
|
3188
|
+
[intrinsic_reference(node), LEAF_PRECEDENCE]
|
|
3189
|
+
when LocalArrayRead
|
|
3190
|
+
cell = local_array_reference(node)
|
|
3191
|
+
widening = widening_cast(node.storage, node.type)
|
|
3192
|
+
if widening
|
|
3193
|
+
["#{widening}#{parenthesize(cell, LEAF_PRECEDENCE, UNARY_PRECEDENCE)}",
|
|
3194
|
+
UNARY_PRECEDENCE]
|
|
3195
|
+
else
|
|
3196
|
+
[cell, LEAF_PRECEDENCE]
|
|
3197
|
+
end
|
|
1864
3198
|
when ElementRead
|
|
1865
3199
|
cell = cell_reference(node.array, node.subscripts)
|
|
1866
|
-
widening = widening_cast(node.array, node.type)
|
|
3200
|
+
widening = widening_cast(array_storage(node.array), node.type)
|
|
1867
3201
|
if widening
|
|
1868
3202
|
["#{widening}#{parenthesize(cell, LEAF_PRECEDENCE, UNARY_PRECEDENCE)}",
|
|
1869
3203
|
UNARY_PRECEDENCE]
|
|
@@ -1871,6 +3205,9 @@ class CArray
|
|
|
1871
3205
|
[cell, LEAF_PRECEDENCE]
|
|
1872
3206
|
end
|
|
1873
3207
|
when MaskTest then emit_mask_test(node)
|
|
3208
|
+
when LocalArrayMaskTest then emit_local_array_mask_test(node)
|
|
3209
|
+
when NumericPredicate then emit_numeric_predicate(node)
|
|
3210
|
+
when Clamp then emit_clamp(node)
|
|
1874
3211
|
when UnaryMinus
|
|
1875
3212
|
operand, precedence = emit_raw(node.operand)
|
|
1876
3213
|
["-#{parenthesize(operand, precedence, UNARY_PRECEDENCE)}", UNARY_PRECEDENCE]
|
|
@@ -1891,7 +3228,11 @@ class CArray
|
|
|
1891
3228
|
when Power then emit_power(node)
|
|
1892
3229
|
when MathCall then emit_math_call(node)
|
|
1893
3230
|
when ArrayAddress
|
|
1894
|
-
[
|
|
3231
|
+
[bare_name(node.array), LEAF_PRECEDENCE]
|
|
3232
|
+
when LocalArrayAddress
|
|
3233
|
+
# The array's own name, which is the address: C decays it at the
|
|
3234
|
+
# call, and there is no buffer for the caller to have packed.
|
|
3235
|
+
[local_c_name(node.name, node.binding), LEAF_PRECEDENCE]
|
|
1895
3236
|
when PointerRead
|
|
1896
3237
|
# A pointer parameter is reached the way C reaches it: contiguous,
|
|
1897
3238
|
# from the address it was handed. No base, no stride, no bounds --
|
|
@@ -1905,7 +3246,8 @@ class CArray
|
|
|
1905
3246
|
# scalars; this one is right here, so the call is direct.
|
|
1906
3247
|
arguments = node.arguments.zip(node.parameters)
|
|
1907
3248
|
.map { |argument, parameter|
|
|
1908
|
-
if argument.is_a?(ArrayAddress)
|
|
3249
|
+
if argument.is_a?(ArrayAddress) ||
|
|
3250
|
+
argument.is_a?(LocalArrayAddress)
|
|
1909
3251
|
emit(argument, :address)
|
|
1910
3252
|
else
|
|
1911
3253
|
emit(argument, parameter.computation)
|
|
@@ -1913,11 +3255,21 @@ class CArray
|
|
|
1913
3255
|
}
|
|
1914
3256
|
arguments << ERROR_FLAG if @error_parameter
|
|
1915
3257
|
["#{@own_symbol}(#{arguments.join(', ')})", LEAF_PRECEDENCE]
|
|
3258
|
+
when RandomDraw
|
|
3259
|
+
# The state is the address of the captured array, which the
|
|
3260
|
+
# declarations above bound to a name; the draw advances it in
|
|
3261
|
+
# place, which is what leaves the generator where the kernel left
|
|
3262
|
+
# it.
|
|
3263
|
+
symbol = CArray::Rng::DRAW_FUNCTIONS
|
|
3264
|
+
.fetch(@randoms.fetch(node.generator).generator)
|
|
3265
|
+
.fetch(node.kind)
|
|
3266
|
+
["#{symbol}(#{bare_name(node.state)})", LEAF_PRECEDENCE]
|
|
1916
3267
|
when CFunctionCall
|
|
1917
3268
|
c_function = @c_functions.fetch(node.name)
|
|
1918
3269
|
arguments = node.arguments.zip(c_function.parameters)
|
|
1919
3270
|
.map { |argument, parameter|
|
|
1920
|
-
if argument.is_a?(ArrayAddress)
|
|
3271
|
+
if argument.is_a?(ArrayAddress) ||
|
|
3272
|
+
argument.is_a?(LocalArrayAddress)
|
|
1921
3273
|
emit(argument, :address)
|
|
1922
3274
|
else
|
|
1923
3275
|
emit(argument, parameter.computation)
|
|
@@ -1925,7 +3277,18 @@ class CArray
|
|
|
1925
3277
|
}
|
|
1926
3278
|
# A pasted body is reached by its symbol; an address by the local
|
|
1927
3279
|
# the declarations bound it to.
|
|
1928
|
-
called = c_function.pasted?
|
|
3280
|
+
called = if c_function.pasted?
|
|
3281
|
+
c_function.symbol
|
|
3282
|
+
elsif @in_function
|
|
3283
|
+
# Borrowed, inside a body that has no `functions` buffer
|
|
3284
|
+
# to take an address in -- so it is called by the name it
|
|
3285
|
+
# has, declared in the preamble, and the linker or the
|
|
3286
|
+
# loader resolves it. `jit_extern` opened the library to
|
|
3287
|
+
# find it, so by here the symbol is in the process.
|
|
3288
|
+
c_function.symbol
|
|
3289
|
+
else
|
|
3290
|
+
bare_name(node.name)
|
|
3291
|
+
end
|
|
1929
3292
|
# And one that can report a failure is handed the slot this kernel
|
|
1930
3293
|
# is watching -- null under a masked cell, where the value written
|
|
1931
3294
|
# is out of contract and a division by zero there was not asked
|
|
@@ -1941,11 +3304,56 @@ class CArray
|
|
|
1941
3304
|
end
|
|
1942
3305
|
|
|
1943
3306
|
# `a[i] == UNDEF` reads the mask byte, never the value.
|
|
3307
|
+
# `Math.gamma` is Ruby's, which is tgamma with two things around it: a
|
|
3308
|
+
# table of exact values for a small integer argument, and the domain
|
|
3309
|
+
# error Ruby raises where tgamma answers with a NaN.
|
|
3310
|
+
def emit_gamma (node)
|
|
3311
|
+
type = node.type
|
|
3312
|
+
unhandled_type(node, "Math.gamma") unless GAMMA_C_TYPES.key?(type)
|
|
3313
|
+
@gamma_types |= [type]
|
|
3314
|
+
suffix, = GAMMA_C_TYPES.fetch(type)
|
|
3315
|
+
["carray_jit_gamma_#{suffix}(#{emit(node.arguments.first, type)}, " \
|
|
3316
|
+
"#{error_argument})", LEAF_PRECEDENCE]
|
|
3317
|
+
end
|
|
3318
|
+
|
|
3319
|
+
# `x.clamp(low, high)` through a helper of its own width, so that each
|
|
3320
|
+
# of the three is computed once and the two failures Ruby raises for
|
|
3321
|
+
# are reported where they happen.
|
|
3322
|
+
def emit_clamp (node)
|
|
3323
|
+
type = node.type
|
|
3324
|
+
unhandled_type(node, "clamp") unless CLAMP_C_TYPES.key?(type)
|
|
3325
|
+
@clamp_types |= [type]
|
|
3326
|
+
["carray_jit_clamp_#{CLAMP_C_TYPES.fetch(type).first}(" \
|
|
3327
|
+
"#{emit(node.value, type)}, #{emit(node.low, type)}, " \
|
|
3328
|
+
"#{emit(node.high, type)}, #{error_argument})", LEAF_PRECEDENCE]
|
|
3329
|
+
end
|
|
3330
|
+
|
|
3331
|
+
# `isnan` and `isfinite` are C's own, and take a number of any width.
|
|
3332
|
+
# A Complex answers `finite?` the way Ruby answers it -- both parts
|
|
3333
|
+
# finite -- through a helper, so that the number is computed once.
|
|
3334
|
+
# An Integer is finite whatever it holds, and goes through `isfinite`
|
|
3335
|
+
# all the same rather than being answered here: the read it stands for
|
|
3336
|
+
# is a read, and dropping it would drop what the read reports.
|
|
3337
|
+
def emit_numeric_predicate (node)
|
|
3338
|
+
type = node.operand.type
|
|
3339
|
+
function = node.name == :nan? ? "isnan" : "isfinite"
|
|
3340
|
+
if complex_type?(type)
|
|
3341
|
+
return emit_helper_call("complex_finite", [emit(node.operand, type)],
|
|
3342
|
+
type == :float_complex)
|
|
3343
|
+
end
|
|
3344
|
+
["#{function}(#{emit(node.operand, type)})", LEAF_PRECEDENCE]
|
|
3345
|
+
end
|
|
3346
|
+
|
|
1944
3347
|
def emit_mask_test (node)
|
|
1945
3348
|
cell = cell_reference(node.array, node.subscripts, :mask)
|
|
1946
3349
|
node.negated ? ["! #{cell}", UNARY_PRECEDENCE] : [cell, LEAF_PRECEDENCE]
|
|
1947
3350
|
end
|
|
1948
3351
|
|
|
3352
|
+
def emit_local_array_mask_test (node)
|
|
3353
|
+
cell = local_array_reference(node, :mask)
|
|
3354
|
+
node.negated ? ["! #{cell}", UNARY_PRECEDENCE] : [cell, LEAF_PRECEDENCE]
|
|
3355
|
+
end
|
|
3356
|
+
|
|
1949
3357
|
# Float#floor and friends hand back an Integer in Ruby, so the C rounds
|
|
1950
3358
|
# and then narrows. An Integer receiver is already there.
|
|
1951
3359
|
def emit_conversion (node)
|
|
@@ -1957,7 +3365,28 @@ class CArray
|
|
|
1957
3365
|
UNARY_PRECEDENCE]
|
|
1958
3366
|
end
|
|
1959
3367
|
return [emit(node.operand, :double), LEAF_PRECEDENCE] if node.result_type == :double
|
|
1960
|
-
|
|
3368
|
+
@uses_whole_number = true
|
|
3369
|
+
["carray_jit_whole_int64(#{node.name}(#{emit(node.operand, :double)}), " \
|
|
3370
|
+
"#{error_argument})", LEAF_PRECEDENCE]
|
|
3371
|
+
end
|
|
3372
|
+
|
|
3373
|
+
# The same rounding where the answer is wanted as a real number: the
|
|
3374
|
+
# Integer Ruby answers is exact however large, and a float cell holds
|
|
3375
|
+
# it as the rounded double already is, so it does not go through int64
|
|
3376
|
+
# and is not held to its range.
|
|
3377
|
+
def rounded_real? (node, target)
|
|
3378
|
+
node.is_a?(Conversion) && node.result_type == :int64 &&
|
|
3379
|
+
node.type == :int64 &&
|
|
3380
|
+
!TypeAssignment::INTEGER_TYPES.include?(node.operand.type) &&
|
|
3381
|
+
[:double, :float].include?(target)
|
|
3382
|
+
end
|
|
3383
|
+
|
|
3384
|
+
def emit_rounded_real (node, target)
|
|
3385
|
+
@uses_whole_number = true
|
|
3386
|
+
text = "carray_jit_whole_real(#{node.name}(#{emit(node.operand, :double)}), " \
|
|
3387
|
+
"#{error_argument})"
|
|
3388
|
+
return [text, LEAF_PRECEDENCE] if target == :double
|
|
3389
|
+
["(float)#{text}", UNARY_PRECEDENCE]
|
|
1961
3390
|
end
|
|
1962
3391
|
|
|
1963
3392
|
# creal and cimag are the way out of the complex type; conj stays in it.
|
|
@@ -2018,7 +3447,7 @@ class CArray
|
|
|
2018
3447
|
LEAF_PRECEDENCE]
|
|
2019
3448
|
end
|
|
2020
3449
|
|
|
2021
|
-
# cabs, fabs and
|
|
3450
|
+
# cabs, fabs and an integer helper are three rather than one because C has
|
|
2022
3451
|
# no generic for them, so this names each type instead of letting one of
|
|
2023
3452
|
# them be what a type it has not heard of falls into.
|
|
2024
3453
|
def emit_absolute_value (node)
|
|
@@ -2033,8 +3462,10 @@ class CArray
|
|
|
2033
3462
|
case node.type
|
|
2034
3463
|
when :double then ["fabs(#{emit(node.operand, :double)})", LEAF_PRECEDENCE]
|
|
2035
3464
|
when :float then ["fabsf(#{emit(node.operand, :float)})", LEAF_PRECEDENCE]
|
|
2036
|
-
when :int64
|
|
2037
|
-
|
|
3465
|
+
when :int64
|
|
3466
|
+
@uses_integer_abs = true
|
|
3467
|
+
["carray_jit_integer_abs(#{emit(node.operand, :int64)})", LEAF_PRECEDENCE]
|
|
3468
|
+
# An unsigned number is its own magnitude, and a signed helper would take it
|
|
2038
3469
|
# through a signed type on the way.
|
|
2039
3470
|
when :uint64 then [emit(node.operand, :uint64), LEAF_PRECEDENCE]
|
|
2040
3471
|
else unhandled_type(node, "abs")
|
|
@@ -2090,10 +3521,13 @@ class CArray
|
|
|
2090
3521
|
# and dividing by one divides each part.
|
|
2091
3522
|
#
|
|
2092
3523
|
# Subtraction is the exception: there the zero really is subtracted, in
|
|
2093
|
-
# Ruby as in C, so `z - x` and `x - z` are C's own operator.
|
|
2094
|
-
#
|
|
2095
|
-
#
|
|
2096
|
-
#
|
|
3524
|
+
# Ruby as in C, so `z - x` and `x - z` are C's own operator. `x * z`
|
|
3525
|
+
# is not the exception it looks like: Ruby coerces the x and multiplies
|
|
3526
|
+
# out in full, as it does two Complex numbers, and a full product in
|
|
3527
|
+
# Ruby is safe_mul's part by part -- a zero times an infinity a zero --
|
|
3528
|
+
# where C's `*` recovers from infinities by Annex G's rules. So both
|
|
3529
|
+
# go through complex.c's comp_mul, and `2.0 * Complex(1.0, -0.0)` and
|
|
3530
|
+
# `Complex(1.0, -0.0) * 2.0` still differ, each reproduced its own way.
|
|
2097
3531
|
#
|
|
2098
3532
|
# And a complex division is Smith's method as complex.c writes it,
|
|
2099
3533
|
# which is not what the C library's __divdc3 computes.
|
|
@@ -2114,18 +3548,7 @@ class CArray
|
|
|
2114
3548
|
if complex_type == :float_complex &&
|
|
2115
3549
|
[:*, :/].include?(node.operator) &&
|
|
2116
3550
|
complex_type?(node.left.type) && complex_type?(node.right.type)
|
|
2117
|
-
|
|
2118
|
-
if helper
|
|
2119
|
-
text, = emit_helper_call(*helper, false)
|
|
2120
|
-
else
|
|
2121
|
-
precedence = PRECEDENCE.fetch(node.operator)
|
|
2122
|
-
# Parenthesised, because the cast binds tighter than the operator:
|
|
2123
|
-
# without them it would narrow the left operand and leave the
|
|
2124
|
-
# multiplication to be worked out around it.
|
|
2125
|
-
text = "(#{emit_operand(node.left, :complex, precedence)} " \
|
|
2126
|
-
"#{node.operator} " \
|
|
2127
|
-
"#{emit_operand(node.right, :complex, precedence)})"
|
|
2128
|
-
end
|
|
3551
|
+
text, = emit_helper_call(*mixed_helper(node, :complex), false)
|
|
2129
3552
|
return ["(float _Complex)#{text}", UNARY_PRECEDENCE]
|
|
2130
3553
|
end
|
|
2131
3554
|
narrow = complex_type == :float_complex
|
|
@@ -2138,6 +3561,9 @@ class CArray
|
|
|
2138
3561
|
end
|
|
2139
3562
|
|
|
2140
3563
|
def emit_helper_call (name, arguments, narrow)
|
|
3564
|
+
COMPLEX_HELPER_NEEDS.fetch(name, []).each do |needed|
|
|
3565
|
+
@complex_helpers |= [[needed, narrow]]
|
|
3566
|
+
end
|
|
2141
3567
|
@complex_helpers |= [[name, narrow]]
|
|
2142
3568
|
prefix = narrow ? "carray_jit_f_" : "carray_jit_"
|
|
2143
3569
|
["#{prefix}#{name}(#{arguments.join(', ')})", LEAF_PRECEDENCE]
|
|
@@ -2168,10 +3594,20 @@ class CArray
|
|
|
2168
3594
|
when :*
|
|
2169
3595
|
if real_type?(right)
|
|
2170
3596
|
["complex_mul_real", [emit(left, ctype), emit(right, rtype)]]
|
|
3597
|
+
elsif real_type?(left)
|
|
3598
|
+
["real_multiply_complex", [emit(left, rtype), emit(right, ctype)]]
|
|
3599
|
+
else
|
|
3600
|
+
["complex_multiply", [emit(left, ctype), emit(right, ctype)]]
|
|
2171
3601
|
end
|
|
2172
3602
|
end
|
|
2173
3603
|
end
|
|
2174
3604
|
|
|
3605
|
+
# The helpers that call another, which has to be emitted above them.
|
|
3606
|
+
COMPLEX_HELPER_NEEDS = {
|
|
3607
|
+
"complex_multiply" => ["safe_multiply"],
|
|
3608
|
+
"real_multiply_complex" => ["safe_multiply"],
|
|
3609
|
+
}.freeze
|
|
3610
|
+
|
|
2175
3611
|
def complex_type? (type)
|
|
2176
3612
|
TypeAssignment.complex?(type)
|
|
2177
3613
|
end
|
|
@@ -2220,6 +3656,7 @@ class CArray
|
|
|
2220
3656
|
return ["(float _Complex)#{text}", UNARY_PRECEDENCE]
|
|
2221
3657
|
end
|
|
2222
3658
|
function = node.name.to_s
|
|
3659
|
+
return emit_gamma(node) if function == "gamma"
|
|
2223
3660
|
function += "f" if node.type == :float
|
|
2224
3661
|
arguments = node.arguments.map { |argument| emit(argument, node.type) }
|
|
2225
3662
|
["#{function}(#{arguments.join(', ')})", LEAF_PRECEDENCE]
|
|
@@ -2272,11 +3709,11 @@ class CArray
|
|
|
2272
3709
|
if node.type == :float
|
|
2273
3710
|
@uses_floor_modulo_float = true
|
|
2274
3711
|
return ["carray_jit_floor_modulo_float(#{emit(node.left, :float)}, " \
|
|
2275
|
-
"#{emit(node.right, :float)})", LEAF_PRECEDENCE]
|
|
3712
|
+
"#{emit(node.right, :float)}, #{error_argument})", LEAF_PRECEDENCE]
|
|
2276
3713
|
end
|
|
2277
3714
|
unhandled_type(node, "%") unless node.type == :double
|
|
2278
3715
|
["carray_jit_floor_modulo_real(#{emit(node.left, :double)}, " \
|
|
2279
|
-
"#{emit(node.right, :double)})", LEAF_PRECEDENCE]
|
|
3716
|
+
"#{emit(node.right, :double)}, #{error_argument})", LEAF_PRECEDENCE]
|
|
2280
3717
|
end
|
|
2281
3718
|
|
|
2282
3719
|
def power_of_two_shift (node)
|
|
@@ -2306,9 +3743,11 @@ class CArray
|
|
|
2306
3743
|
text
|
|
2307
3744
|
end
|
|
2308
3745
|
|
|
3746
|
+
# Spelled with `__`, which is what keeps it out of a block's way: a
|
|
3747
|
+
# local with `__` in its name is moved (see `local_c_names`).
|
|
2309
3748
|
def next_temporary (prefix = "result")
|
|
2310
3749
|
@temporary_count += 1
|
|
2311
|
-
"#{prefix}#{@temporary_count}"
|
|
3750
|
+
"#{prefix}__#{@temporary_count}"
|
|
2312
3751
|
end
|
|
2313
3752
|
|
|
2314
3753
|
end
|