carray-jit 0.1.2 → 0.1.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +771 -3
- data/README.md +7 -6
- data/carray-jit.gemspec +1 -3
- data/docs/00_Introduction.md +4 -3
- data/docs/01_GettingStarted.md +1 -1
- data/docs/02_KernelShapes.md +93 -14
- data/docs/03_SupportedFeatures.md +582 -26
- data/docs/04_Compiling.md +33 -6
- data/docs/05_DesignNotes.md +3 -3
- data/docs/06_Cheatsheet.md +198 -5
- data/docs/07_StepByStep.ja.md +534 -0
- data/docs/07_StepByStep.md +535 -0
- data/examples/README.md +12 -0
- data/examples/applications/alarm.rb +121 -0
- data/examples/applications/collatz.rb +105 -0
- data/examples/applications/cubic_spline.rb +331 -0
- data/examples/applications/dithering.rb +144 -0
- data/examples/applications/group_stats.rb +115 -0
- data/examples/applications/lookup.rb +126 -0
- data/examples/applications/median_filter.rb +153 -0
- data/examples/applications/parcel_ascent.rb +220 -0
- data/examples/applications/point_in_polygon.rb +111 -0
- data/examples/applications/random_walk.rb +98 -0
- data/examples/applications/van_der_pol.rb +186 -0
- data/examples/applications/wet_bulb.rb +140 -0
- data/examples/features/10_complex.rb +14 -4
- data/examples/features/15_loops.rb +7 -1
- data/lib/carray/jit/access.rb +14 -0
- data/lib/carray/jit/analyzer.rb +2077 -136
- data/lib/carray/jit/block_reader.rb +37 -6
- data/lib/carray/jit/c_function.rb +613 -76
- data/lib/carray/jit/c_generator.rb +1595 -156
- data/lib/carray/jit/call.rb +68 -0
- data/lib/carray/jit/compiler.rb +75 -11
- data/lib/carray/jit/kernel.rb +369 -32
- data/lib/carray/jit/node.rb +359 -9
- data/lib/carray/jit/sorting_networks.rb +182 -0
- data/lib/carray/jit/type_assignment.rb +371 -34
- data/lib/carray/jit/version.rb +1 -1
- data/lib/carray/jit.rb +560 -64
- metadata +22 -8
- data/ext/carray_jit_access/carray_jit_access.c +0 -460
- data/ext/carray_jit_access/extconf.rb +0 -8
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
# Collatz: how long each number takes to fall to 1.
|
|
2
|
+
#
|
|
3
|
+
# Halve it when it is even, take 3n+1 when it is odd, and count the steps
|
|
4
|
+
# until it reaches 1. How many steps that is depends on the number, and
|
|
5
|
+
# nothing about the number says in advance -- 3 takes 7 steps and 27 takes
|
|
6
|
+
# 111 -- so the length of the loop is the cell's own business and is what
|
|
7
|
+
# `while` is for.
|
|
8
|
+
#
|
|
9
|
+
# Written over whole arrays this is a pass per step over everything still
|
|
10
|
+
# running, and every cell pays for the longest one; there is no expression
|
|
11
|
+
# that lets one cell stop. Here each cell stops when it reaches 1, and the
|
|
12
|
+
# two answers -- the step count and the highest value on the way -- are
|
|
13
|
+
# written in the same pass.
|
|
14
|
+
#
|
|
15
|
+
# ruby examples/applications/collatz.rb
|
|
16
|
+
|
|
17
|
+
$LOAD_PATH.unshift(File.expand_path("../../lib", __dir__))
|
|
18
|
+
require "carray/jit"
|
|
19
|
+
|
|
20
|
+
N = 1_000_000
|
|
21
|
+
|
|
22
|
+
def collatz (n)
|
|
23
|
+
steps = CArray.int32(n)
|
|
24
|
+
peak = CArray.int64(n)
|
|
25
|
+
kernel = CArray.jit_for(1...n) { |i|
|
|
26
|
+
x = i
|
|
27
|
+
hi = x
|
|
28
|
+
count = 0
|
|
29
|
+
while x != 1
|
|
30
|
+
x = x % 2 == 0 ? x / 2 : 3 * x + 1
|
|
31
|
+
hi = x if x > hi
|
|
32
|
+
count = count + 1
|
|
33
|
+
end
|
|
34
|
+
steps[i] = count
|
|
35
|
+
peak[i] = hi
|
|
36
|
+
}
|
|
37
|
+
[steps, peak, kernel]
|
|
38
|
+
end
|
|
39
|
+
|
|
40
|
+
steps, peak, kernel = collatz(N)
|
|
41
|
+
|
|
42
|
+
# The numbers that took longer than anything before them.
|
|
43
|
+
puts "record holders below #{N}"
|
|
44
|
+
record = 0
|
|
45
|
+
(1...N).each do |i|
|
|
46
|
+
next unless steps[i] > record
|
|
47
|
+
record = steps[i]
|
|
48
|
+
next unless record > 400
|
|
49
|
+
puts format(" %7d %3d steps, reaching %d", i, record, peak[i])
|
|
50
|
+
end
|
|
51
|
+
puts format(" the highest value reached by any of them is %d, from %d",
|
|
52
|
+
peak.max, peak.max_addr)
|
|
53
|
+
|
|
54
|
+
# `x` is an int64 in the generated C, and nothing here asked for that. A
|
|
55
|
+
# local's type is settled from the whole body: `hi` is what `peak` is given,
|
|
56
|
+
# `peak` is an int64 array, and `hi = x` carries that back to `x`. Which is
|
|
57
|
+
# why 704511 reaches 56 billion without overflowing an int32 on the way.
|
|
58
|
+
puts
|
|
59
|
+
puts "what the local was compiled as"
|
|
60
|
+
puts format(" %s", kernel.c_source.lines.grep(/^\s*int\d+_t x;/).first.strip)
|
|
61
|
+
|
|
62
|
+
# The same walk in Ruby, for the answer and the time. Over the first tenth,
|
|
63
|
+
# because the point of the comparison is the ratio.
|
|
64
|
+
CHECK = N / 10
|
|
65
|
+
|
|
66
|
+
def collatz_in_ruby (n)
|
|
67
|
+
steps = Array.new(n, 0)
|
|
68
|
+
(1...n).each do |i|
|
|
69
|
+
x = i
|
|
70
|
+
count = 0
|
|
71
|
+
while x != 1
|
|
72
|
+
x = x.even? ? x / 2 : 3 * x + 1
|
|
73
|
+
count += 1
|
|
74
|
+
end
|
|
75
|
+
steps[i] = count
|
|
76
|
+
end
|
|
77
|
+
steps
|
|
78
|
+
end
|
|
79
|
+
|
|
80
|
+
in_ruby = collatz_in_ruby(CHECK)
|
|
81
|
+
puts
|
|
82
|
+
puts format("agrees with Ruby over 1...%d %s",
|
|
83
|
+
CHECK, steps[0...CHECK].to_a == in_ruby)
|
|
84
|
+
|
|
85
|
+
collatz(CHECK) # compiled and cached on the first call
|
|
86
|
+
|
|
87
|
+
started = Process.clock_gettime(Process::CLOCK_MONOTONIC)
|
|
88
|
+
5.times { collatz(CHECK) }
|
|
89
|
+
compiled = (Process.clock_gettime(Process::CLOCK_MONOTONIC) - started) / 5
|
|
90
|
+
|
|
91
|
+
started = Process.clock_gettime(Process::CLOCK_MONOTONIC)
|
|
92
|
+
collatz_in_ruby(CHECK)
|
|
93
|
+
interpreted = Process.clock_gettime(Process::CLOCK_MONOTONIC) - started
|
|
94
|
+
|
|
95
|
+
puts format("%d starting values: %.0f ms compiled, %.0f ms in Ruby (%.0fx)",
|
|
96
|
+
CHECK, compiled * 1e3, interpreted * 1e3, interpreted / compiled)
|
|
97
|
+
|
|
98
|
+
# And the loop the kernel runs is the one the block states: `x.even?` is not
|
|
99
|
+
# in the subset, so the parity test is written as the arithmetic it is.
|
|
100
|
+
begin
|
|
101
|
+
CArray.jit_for(1) { |i| steps[i] = i.even? ? 0 : 1 }
|
|
102
|
+
rescue CArray::JIT::Unsupported => error
|
|
103
|
+
puts
|
|
104
|
+
puts "refused: #{error.message.sub(/ \(at line.*/m, "")}"
|
|
105
|
+
end
|
|
@@ -0,0 +1,331 @@
|
|
|
1
|
+
# A cubic spline through measured points, natural and clamped
|
|
2
|
+
#
|
|
3
|
+
# Fitting a C2 curve through n points is a tridiagonal solve: the second
|
|
4
|
+
# derivatives at the knots are the unknowns, and each interior knot gives one
|
|
5
|
+
# row with three entries in it. The rows are diagonally dominant, so the
|
|
6
|
+
# Thomas algorithm applies -- two sequential sweeps, which is the part an
|
|
7
|
+
# array library cannot do for you.
|
|
8
|
+
#
|
|
9
|
+
# The boundary conditions differ only in the first and last row. Natural
|
|
10
|
+
# asks for zero curvature at the ends, and needs nothing but the samples;
|
|
11
|
+
# clamped asks for a given slope there, and needs to be told what it is. What
|
|
12
|
+
# that buys is visible below: where the true curve is still bending at the end,
|
|
13
|
+
# the natural spline flattens it, and the error there is several times what it
|
|
14
|
+
# is anywhere inside -- the clamped fit removes it. Not-a-knot asks for the
|
|
15
|
+
# first two pieces to be one cubic, and the last two as well, which costs no
|
|
16
|
+
# information at all and recovers a good part of the difference.
|
|
17
|
+
#
|
|
18
|
+
# Evaluating the result is a second kernel, and a different shape of one --
|
|
19
|
+
# every query point searches for its interval on its own, so the body is a
|
|
20
|
+
# bisection with the bound in the extent. That search is where the time goes,
|
|
21
|
+
# and it is not always needed: query points that arrive sorted -- resampling
|
|
22
|
+
# onto a grid gives that -- let the interval be carried forward instead of
|
|
23
|
+
# looked up, which is a sweep of the kind the solver already is. Both are
|
|
24
|
+
# here, and they agree bit for bit.
|
|
25
|
+
#
|
|
26
|
+
# ruby examples/applications/cubic_spline.rb
|
|
27
|
+
|
|
28
|
+
$LOAD_PATH.unshift(File.expand_path("../../lib", __dir__))
|
|
29
|
+
require "carray/jit"
|
|
30
|
+
|
|
31
|
+
# The curve being sampled, and its slope -- known here so that the clamped
|
|
32
|
+
# ends have something to be given, and so the error can be measured.
|
|
33
|
+
def curve (x) Math.exp(-0.35 * x) * Math.sin(2.0 * x) end
|
|
34
|
+
def slope (x) Math.exp(-0.35 * x) * (2.0 * Math.cos(2.0 * x) - 0.35 * Math.sin(2.0 * x)) end
|
|
35
|
+
|
|
36
|
+
# ------------------------------------------------------------ the two kernels
|
|
37
|
+
|
|
38
|
+
# The moments M[i] = S''(x[i]). Interior rows come from the continuity of the
|
|
39
|
+
# first derivative; the two end rows are the boundary condition and are the
|
|
40
|
+
# only thing the three of them disagree about.
|
|
41
|
+
#
|
|
42
|
+
# Not-a-knot is the one that does not fit in a row of its own: asking S''' to
|
|
43
|
+
# be continuous at x[1] puts M[0], M[1] and M[2] in one equation, which is a
|
|
44
|
+
# band too wide. Stated the other way round it says M is linear across x[1],
|
|
45
|
+
# so M[0] is an extrapolation of the two inside it -- drop the first and last
|
|
46
|
+
# unknown, solve the n-2 rows between them, and put the two back afterwards.
|
|
47
|
+
# That is what `first_row` and `last_row` are: the same two sweeps, over the
|
|
48
|
+
# rows that are actually unknown.
|
|
49
|
+
def moments (x, y, n, work, ends)
|
|
50
|
+
a, b, c, d, cc, dd, moment = work
|
|
51
|
+
|
|
52
|
+
CArray.jit_for(1...(n-1)) { |i|
|
|
53
|
+
left = x[i] - x[i-1]
|
|
54
|
+
right = x[i+1] - x[i]
|
|
55
|
+
a[i] = left
|
|
56
|
+
b[i] = 2.0 * (left + right)
|
|
57
|
+
c[i] = right
|
|
58
|
+
d[i] = 6.0 * ((y[i+1] - y[i]) / right - (y[i] - y[i-1]) / left)
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
case ends
|
|
62
|
+
when nil # natural: S'' = 0 at both ends
|
|
63
|
+
first_row, last_row = 0, n-1
|
|
64
|
+
b[0] = 1.0 ; c[0] = 0.0 ; d[0] = 0.0
|
|
65
|
+
a[n-1] = 0.0 ; b[n-1] = 1.0 ; d[n-1] = 0.0
|
|
66
|
+
when :not_a_knot # not-a-knot: S''' continuous at x[1], x[n-2]
|
|
67
|
+
raise ArgumentError, "not-a-knot needs at least 4 points, got #{n}" if n < 4
|
|
68
|
+
first_row, last_row = 1, n-2
|
|
69
|
+
h0 = x[1] - x[0] ; h1 = x[2] - x[1] # d[1] and d[n-2] are the interior
|
|
70
|
+
b[1] = 3.0 * h0 + 2.0 * h1 + h0 * h0 / h1 # right-hand sides already
|
|
71
|
+
c[1] = (h1 * h1 - h0 * h0) / h1
|
|
72
|
+
hm = x[n-2] - x[n-3] ; hn = x[n-1] - x[n-2]
|
|
73
|
+
a[n-2] = (hm * hm - hn * hn) / hm
|
|
74
|
+
b[n-2] = 2.0 * hm + 3.0 * hn + hn * hn / hm
|
|
75
|
+
else # clamped: S' given at both ends
|
|
76
|
+
first_row, last_row = 0, n-1
|
|
77
|
+
first, last = ends
|
|
78
|
+
h0 = x[1] - x[0]
|
|
79
|
+
b[0] = 2.0 * h0 ; c[0] = h0
|
|
80
|
+
d[0] = 6.0 * ((y[1] - y[0]) / h0 - first)
|
|
81
|
+
hn = x[n-1] - x[n-2]
|
|
82
|
+
a[n-1] = hn ; b[n-1] = 2.0 * hn
|
|
83
|
+
d[n-1] = 6.0 * (last - (y[n-1] - y[n-2]) / hn)
|
|
84
|
+
end
|
|
85
|
+
|
|
86
|
+
cc[first_row] = c[first_row] / b[first_row] # Thomas, forward
|
|
87
|
+
dd[first_row] = d[first_row] / b[first_row]
|
|
88
|
+
CArray.jit_for((first_row+1)..last_row) { |i|
|
|
89
|
+
denominator = b[i] - a[i] * cc[i-1]
|
|
90
|
+
cc[i] = c[i] / denominator
|
|
91
|
+
dd[i] = (d[i] - a[i] * dd[i-1]) / denominator
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
moment[last_row] = dd[last_row] # Thomas, back substitution
|
|
95
|
+
CArray.jit_for((last_row-1).step(first_row, -1)) { |i|
|
|
96
|
+
moment[i] = dd[i] - cc[i] * moment[i+1]
|
|
97
|
+
}
|
|
98
|
+
|
|
99
|
+
if ends == :not_a_knot # the rows that were left out
|
|
100
|
+
moment[0] = moment[1] - (x[1] - x[0]) * (moment[2] - moment[1]) / (x[2] - x[1])
|
|
101
|
+
moment[n-1] = moment[n-2] + (x[n-1] - x[n-2]) * (moment[n-2] - moment[n-3]) / (x[n-2] - x[n-3])
|
|
102
|
+
end
|
|
103
|
+
moment
|
|
104
|
+
end
|
|
105
|
+
|
|
106
|
+
# Evaluate at arbitrary points. The knots are not equally spaced, so each
|
|
107
|
+
# query point has to find its interval; `passes` bounds the bisection, and a
|
|
108
|
+
# kernel that cannot fail to stop is the better kind.
|
|
109
|
+
def evaluate (x, y, moment, n, query, value, derivative)
|
|
110
|
+
points = query.dim[0]
|
|
111
|
+
passes = Math.log2(n).ceil + 1
|
|
112
|
+
|
|
113
|
+
CArray.jit_for(points) { |k|
|
|
114
|
+
lo = 0
|
|
115
|
+
hi = n - 2
|
|
116
|
+
(0...passes).each { |pass|
|
|
117
|
+
break if lo >= hi
|
|
118
|
+
middle = (lo + hi + 1) / 2
|
|
119
|
+
if x[middle] <= query[k]
|
|
120
|
+
lo = middle
|
|
121
|
+
else
|
|
122
|
+
hi = middle - 1
|
|
123
|
+
end
|
|
124
|
+
}
|
|
125
|
+
h = x[lo+1] - x[lo]
|
|
126
|
+
t = query[k] - x[lo]
|
|
127
|
+
linear = (y[lo+1] - y[lo]) / h - h * (2.0 * moment[lo] + moment[lo+1]) / 6.0
|
|
128
|
+
cubic = (moment[lo+1] - moment[lo]) / (6.0 * h)
|
|
129
|
+
value[k] = y[lo] + t * (linear + t * (0.5 * moment[lo] + t * cubic))
|
|
130
|
+
derivative[k] = linear + t * (moment[lo] + t * 3.0 * cubic)
|
|
131
|
+
}
|
|
132
|
+
end
|
|
133
|
+
|
|
134
|
+
# The same evaluation where the query points are sorted. Then nothing has to
|
|
135
|
+
# search: the interval only moves forward, so one sweep carries it from cell to
|
|
136
|
+
# cell and passes each knot once -- O(m + n) against O(m log n). How far it
|
|
137
|
+
# advances at one cell is the data's business, which is what `while` is for;
|
|
138
|
+
# there is no bound to put in an extent.
|
|
139
|
+
#
|
|
140
|
+
# The first cell has no k-1 to read, so it is seeded here and the sweep starts
|
|
141
|
+
# at 1 -- `jit_for(points)` over a body reading interval[k-1] is refused, and
|
|
142
|
+
# says so, exactly as the Thomas sweeps above are stated from 1.
|
|
143
|
+
#
|
|
144
|
+
# The intervals are found first and evaluated second, which is what keeps the
|
|
145
|
+
# seed from needing a copy of the polynomial: it is one entry in `interval`,
|
|
146
|
+
# and the second kernel does every point the same way. Fusing the two into a
|
|
147
|
+
# single kernel is worth about 30% more, at that price.
|
|
148
|
+
def resample (x, y, moment, n, query, value, derivative, interval)
|
|
149
|
+
points = query.dim[0]
|
|
150
|
+
|
|
151
|
+
seed = 0
|
|
152
|
+
seed += 1 while seed < n - 2 && x[seed+1] <= query[0]
|
|
153
|
+
interval[0] = seed
|
|
154
|
+
|
|
155
|
+
CArray.jit_for(1...points) { |k|
|
|
156
|
+
lo = interval[k-1]
|
|
157
|
+
while lo < n - 2 && x[lo+1] <= query[k]
|
|
158
|
+
lo = lo + 1
|
|
159
|
+
end
|
|
160
|
+
interval[k] = lo
|
|
161
|
+
}
|
|
162
|
+
|
|
163
|
+
CArray.jit_for(points) { |k|
|
|
164
|
+
lo = interval[k]
|
|
165
|
+
h = x[lo+1] - x[lo]
|
|
166
|
+
t = query[k] - x[lo]
|
|
167
|
+
linear = (y[lo+1] - y[lo]) / h - h * (2.0 * moment[lo] + moment[lo+1]) / 6.0
|
|
168
|
+
cubic = (moment[lo+1] - moment[lo]) / (6.0 * h)
|
|
169
|
+
value[k] = y[lo] + t * (linear + t * (0.5 * moment[lo] + t * cubic))
|
|
170
|
+
derivative[k] = linear + t * (moment[lo] + t * 3.0 * cubic)
|
|
171
|
+
}
|
|
172
|
+
end
|
|
173
|
+
|
|
174
|
+
# ------------------------------------------------------------------ the fit
|
|
175
|
+
|
|
176
|
+
n = 15
|
|
177
|
+
random = Random.new(20260909)
|
|
178
|
+
# Knots that are not equally spaced -- measurements rarely are, and nothing
|
|
179
|
+
# above assumed they would be.
|
|
180
|
+
x = CArray.double(n) { |i| 6.0 * i / (n - 1) }
|
|
181
|
+
(1...(n-1)).each { |i| x[i] += random.rand(-0.12..0.12) }
|
|
182
|
+
y = CArray.double(n) { |i| curve(x[i]) }
|
|
183
|
+
|
|
184
|
+
work = Array.new(7) { CArray.double(n) }
|
|
185
|
+
natural = moments(x, y, n, work, nil).copy
|
|
186
|
+
clamped = moments(x, y, n, work, [slope(x[0]), slope(x[n-1])]).copy
|
|
187
|
+
not_a_knot = moments(x, y, n, work, :not_a_knot).copy
|
|
188
|
+
|
|
189
|
+
points = 601
|
|
190
|
+
query = CArray.double(points) { |k| x[0] + (x[n-1] - x[0]) * k / (points - 1) }
|
|
191
|
+
truth = CArray.double(points) { |k| curve(query[k]) }
|
|
192
|
+
|
|
193
|
+
value = CArray.double(points)
|
|
194
|
+
derivative = CArray.double(points)
|
|
195
|
+
|
|
196
|
+
puts "cubic spline through #{n} unevenly spaced points, sampled at #{points}"
|
|
197
|
+
|
|
198
|
+
results = {}
|
|
199
|
+
{ "natural" => natural, "clamped" => clamped, "not-a-knot" => not_a_knot }.each do |name, moment|
|
|
200
|
+
evaluate(x, y, moment, n, query, value, derivative)
|
|
201
|
+
results[name] = [value.copy, derivative.copy]
|
|
202
|
+
|
|
203
|
+
error = (value - truth).abs
|
|
204
|
+
edge = x[1] - x[0] # the first and last interval
|
|
205
|
+
ends = (0...points).select { |k| query[k] < x[0] + edge || query[k] > x[n-1] - edge }
|
|
206
|
+
inside = (0...points).to_a - ends
|
|
207
|
+
|
|
208
|
+
puts format(" %-10s max error %.2e overall, %.2e in the end intervals, %.2e inside",
|
|
209
|
+
name, error.max, ends.map { |k| error[k] }.max, inside.map { |k| error[k] }.max)
|
|
210
|
+
end
|
|
211
|
+
|
|
212
|
+
# What each boundary condition asked for, checked at the ends.
|
|
213
|
+
puts format(" natural S''(a) = %.1e, S''(b) = %.1e -- zero, by construction",
|
|
214
|
+
natural[0], natural[n-1])
|
|
215
|
+
puts format(" clamped S'(a) = %+.6f vs %+.6f asked for", results["clamped"][1][0], slope(x[0]))
|
|
216
|
+
puts format(" S'(b) = %+.6f vs %+.6f", results["clamped"][1][points-1], slope(x[n-1]))
|
|
217
|
+
jump = lambda { |m, i, j, k| # S''' across the knot that is not one
|
|
218
|
+
(m[j] - m[i]) / (x[j] - x[i]) - (m[k] - m[j]) / (x[k] - x[j])
|
|
219
|
+
}
|
|
220
|
+
puts format(" not-knot S''' jumps by %.1e at x[1] and %.1e at x[n-2] -- neither is a knot",
|
|
221
|
+
jump.call(not_a_knot, 0, 1, 2), jump.call(not_a_knot, n-3, n-2, n-1))
|
|
222
|
+
|
|
223
|
+
# The interpolation itself: all of them pass through every knot, so all three
|
|
224
|
+
# are asked, not just the one whose name comes first.
|
|
225
|
+
knots = CArray.double(n) { |i| x[i] }
|
|
226
|
+
at_knots = CArray.double(n)
|
|
227
|
+
at_knots_slope = CArray.double(n)
|
|
228
|
+
{ "natural" => natural, "clamped" => clamped, "not-a-knot" => not_a_knot }.each do |name, moment|
|
|
229
|
+
evaluate(x, y, moment, n, knots, at_knots, at_knots_slope)
|
|
230
|
+
puts format(" %-10s max |S(x_i) - y_i| = %.2e", name, (at_knots - y).abs.max)
|
|
231
|
+
end
|
|
232
|
+
|
|
233
|
+
# The query grid above is sorted, so the sweep applies to it -- and gives back
|
|
234
|
+
# the same doubles, not merely close ones: the interval a point lands in is the
|
|
235
|
+
# same interval, and the polynomial evaluated in it is the same expression.
|
|
236
|
+
swept = CArray.double(points)
|
|
237
|
+
swept_slope = CArray.double(points)
|
|
238
|
+
resample(x, y, clamped, n, query, swept, swept_slope, CArray.int32(points))
|
|
239
|
+
puts " the sorted sweep agrees bit for bit #{swept.to_a == results["clamped"][0].to_a}"
|
|
240
|
+
|
|
241
|
+
# The curve, and the samples it was built from.
|
|
242
|
+
rows, columns = 15, 74
|
|
243
|
+
low, high = -0.55, 0.85
|
|
244
|
+
canvas = Array.new(rows) { " " * columns }
|
|
245
|
+
plot = lambda { |xs, ys, mark|
|
|
246
|
+
xs.each_with_index do |xv, k|
|
|
247
|
+
column = ((xv - x[0]) / (x[n-1] - x[0]) * (columns - 1)).round
|
|
248
|
+
row = ((high - ys[k]) / (high - low) * (rows - 1)).round
|
|
249
|
+
canvas[row][column] = mark if row.between?(0, rows - 1) && column.between?(0, columns - 1)
|
|
250
|
+
end
|
|
251
|
+
}
|
|
252
|
+
plot.call(query.to_a, results["clamped"][0].to_a, ".")
|
|
253
|
+
plot.call(x.to_a, y.to_a, "o")
|
|
254
|
+
puts
|
|
255
|
+
canvas.each { |row| puts " |#{row}|" }
|
|
256
|
+
puts " o samples, . the clamped spline through them"
|
|
257
|
+
|
|
258
|
+
# ------------------------------------------------------------------- the cost
|
|
259
|
+
|
|
260
|
+
def time (repeats)
|
|
261
|
+
started = Process.clock_gettime(Process::CLOCK_MONOTONIC)
|
|
262
|
+
repeats.times { yield }
|
|
263
|
+
(Process.clock_gettime(Process::CLOCK_MONOTONIC) - started) / repeats
|
|
264
|
+
end
|
|
265
|
+
|
|
266
|
+
# The same two kernels written as Ruby loops. The fit is O(n) and the
|
|
267
|
+
# evaluation O(m log n), so both grow slowly -- what the compiled version
|
|
268
|
+
# removes is the per-element cost, which is where all of the difference is.
|
|
269
|
+
def ruby_moments (x, y, n, work)
|
|
270
|
+
a, b, c, d, cc, dd, moment = work
|
|
271
|
+
(1...(n-1)).each do |i|
|
|
272
|
+
left = x[i] - x[i-1]
|
|
273
|
+
right = x[i+1] - x[i]
|
|
274
|
+
a[i] = left
|
|
275
|
+
b[i] = 2.0 * (left + right)
|
|
276
|
+
c[i] = right
|
|
277
|
+
d[i] = 6.0 * ((y[i+1] - y[i]) / right - (y[i] - y[i-1]) / left)
|
|
278
|
+
end
|
|
279
|
+
b[0] = 1.0 ; c[0] = 0.0 ; d[0] = 0.0
|
|
280
|
+
a[n-1] = 0.0 ; b[n-1] = 1.0 ; d[n-1] = 0.0
|
|
281
|
+
cc[0] = c[0] / b[0]
|
|
282
|
+
dd[0] = d[0] / b[0]
|
|
283
|
+
(1...n).each do |i|
|
|
284
|
+
denominator = b[i] - a[i] * cc[i-1]
|
|
285
|
+
cc[i] = c[i] / denominator
|
|
286
|
+
dd[i] = (d[i] - a[i] * dd[i-1]) / denominator
|
|
287
|
+
end
|
|
288
|
+
moment[n-1] = dd[n-1]
|
|
289
|
+
(n-2).step(0, -1) do |i|
|
|
290
|
+
moment[i] = dd[i] - cc[i] * moment[i+1]
|
|
291
|
+
end
|
|
292
|
+
moment
|
|
293
|
+
end
|
|
294
|
+
|
|
295
|
+
def ruby_evaluate (x, y, moment, n, query, value, derivative)
|
|
296
|
+
query.dim[0].times do |k|
|
|
297
|
+
lo, hi = 0, n - 2
|
|
298
|
+
while lo < hi
|
|
299
|
+
middle = (lo + hi + 1) / 2
|
|
300
|
+
if x[middle] <= query[k] then lo = middle else hi = middle - 1 end
|
|
301
|
+
end
|
|
302
|
+
h = x[lo+1] - x[lo]
|
|
303
|
+
t = query[k] - x[lo]
|
|
304
|
+
linear = (y[lo+1] - y[lo]) / h - h * (2.0 * moment[lo] + moment[lo+1]) / 6.0
|
|
305
|
+
cubic = (moment[lo+1] - moment[lo]) / (6.0 * h)
|
|
306
|
+
value[k] = y[lo] + t * (linear + t * (0.5 * moment[lo] + t * cubic))
|
|
307
|
+
derivative[k] = linear + t * (moment[lo] + t * 3.0 * cubic)
|
|
308
|
+
end
|
|
309
|
+
end
|
|
310
|
+
|
|
311
|
+
puts
|
|
312
|
+
[[200, 20_000], [2_000, 200_000]].each do |size, sampled|
|
|
313
|
+
knots = CArray.double(size) { |i| 6.0 * i / (size - 1) }
|
|
314
|
+
values = CArray.double(size) { |i| curve(knots[i]) }
|
|
315
|
+
scratch = Array.new(7) { CArray.double(size) }
|
|
316
|
+
at = CArray.double(sampled) { |k| 6.0 * k / (sampled - 1) }
|
|
317
|
+
out, slopes = CArray.double(sampled), CArray.double(sampled)
|
|
318
|
+
|
|
319
|
+
fit = time(20) { moments(knots, values, size, scratch, nil) }
|
|
320
|
+
fit_ruby = time(3) { ruby_moments(knots, values, size, scratch) }
|
|
321
|
+
moment = moments(knots, values, size, scratch, nil)
|
|
322
|
+
sample = time(5) { evaluate(knots, values, moment, size, at, out, slopes) }
|
|
323
|
+
sample_ruby = time(2) { ruby_evaluate(knots, values, moment, size, at, out, slopes) }
|
|
324
|
+
cells = CArray.int32(sampled)
|
|
325
|
+
swept = time(5) { resample(knots, values, moment, size, at, out, slopes, cells) }
|
|
326
|
+
|
|
327
|
+
puts format(" n = %5d fit %7.1f us vs %8.1f us Ruby (%3.0fx)", size, fit * 1e6, fit_ruby * 1e6, fit_ruby / fit)
|
|
328
|
+
puts format(" m = %6d eval %7.1f us vs %8.1f us Ruby (%3.0fx)", sampled, sample * 1e6, sample_ruby * 1e6, sample_ruby / sample)
|
|
329
|
+
puts format(" sorted %7.1f us -- the same points, with the search taken out (%.1fx)",
|
|
330
|
+
swept * 1e6, sample / swept)
|
|
331
|
+
end
|
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
# Floyd-Steinberg dithering: one bit per pixel, and the error passed on.
|
|
2
|
+
#
|
|
3
|
+
# Each pixel is rounded to black or white, and what the rounding threw away is
|
|
4
|
+
# handed to the neighbours that have not been visited yet -- seven sixteenths
|
|
5
|
+
# to the right, and the rest to the row below. So a cell writes the cells the
|
|
6
|
+
# loop is about to read, and the order it visits them in is not an
|
|
7
|
+
# optimisation but the definition: run the same rule right to left and a
|
|
8
|
+
# different picture comes out.
|
|
9
|
+
#
|
|
10
|
+
# That is what separates this from `sobel_edges.rb`, where every cell only
|
|
11
|
+
# reads its neighbours and the pass is a stencil. Here there is no window, no
|
|
12
|
+
# expression over whole arrays, and no order to be derived -- there is a walk,
|
|
13
|
+
# and the walk is the algorithm. The kernel is one cell whose body is that
|
|
14
|
+
# walk, which is the shape to reach for when the order is the point.
|
|
15
|
+
#
|
|
16
|
+
# ruby examples/applications/dithering.rb
|
|
17
|
+
|
|
18
|
+
$LOAD_PATH.unshift(File.expand_path("../../lib", __dir__))
|
|
19
|
+
require "carray/jit"
|
|
20
|
+
|
|
21
|
+
# A grey ramp, top to bottom, with a brighter disc sitting in it.
|
|
22
|
+
def picture (rows, columns)
|
|
23
|
+
image = CArray.double(rows, columns)
|
|
24
|
+
CArray.jit_for(rows, columns) { |y, x|
|
|
25
|
+
dy = (y - rows / 2.0) / (rows / 2.0)
|
|
26
|
+
dx = (x - columns / 2.0) / (columns / 2.0) * 0.42
|
|
27
|
+
ramp = 0.08 + 0.84 * y / rows
|
|
28
|
+
image[y, x] = dx * dx + dy * dy < 0.16 ? ramp * 0.35 + 0.62 : ramp
|
|
29
|
+
}
|
|
30
|
+
image
|
|
31
|
+
end
|
|
32
|
+
|
|
33
|
+
# The work array carries a row below and a column on each side, so that the
|
|
34
|
+
# four neighbours a pixel writes are always cells that exist. A kernel's
|
|
35
|
+
# bounds are checked against the subscripts it is written with rather than
|
|
36
|
+
# against the branches that guard them, so the room is made in the array, not
|
|
37
|
+
# in an `if`.
|
|
38
|
+
def dither (image)
|
|
39
|
+
rows, columns = image.dim
|
|
40
|
+
out = CArray.int8(rows, columns)
|
|
41
|
+
work = CArray.double(rows + 1, columns + 2)
|
|
42
|
+
work[0..-2, 1..-2] = image
|
|
43
|
+
CArray.jit_for(1) { |z|
|
|
44
|
+
(0...rows).each { |y|
|
|
45
|
+
(0...columns).each { |x|
|
|
46
|
+
old = work[y, x + 1]
|
|
47
|
+
new = old > 0.5 ? 1.0 : 0.0
|
|
48
|
+
out[y, x] = new
|
|
49
|
+
error = old - new
|
|
50
|
+
work[y, x + 2] += error * 7.0 / 16.0
|
|
51
|
+
work[y + 1, x] += error * 3.0 / 16.0
|
|
52
|
+
work[y + 1, x + 1] += error * 5.0 / 16.0
|
|
53
|
+
work[y + 1, x + 2] += error * 1.0 / 16.0
|
|
54
|
+
}
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
out
|
|
58
|
+
end
|
|
59
|
+
|
|
60
|
+
ROWS, COLUMNS = 22, 72
|
|
61
|
+
|
|
62
|
+
def show (bits, title)
|
|
63
|
+
puts title
|
|
64
|
+
bits.dim[0].times do |y|
|
|
65
|
+
puts " " + (0...bits.dim[1]).map { |x| bits[y, x] == 1 ? "@" : " " }.join
|
|
66
|
+
end
|
|
67
|
+
puts
|
|
68
|
+
end
|
|
69
|
+
|
|
70
|
+
image = picture(ROWS, COLUMNS)
|
|
71
|
+
|
|
72
|
+
# Rounding each pixel on its own is an expression over the whole array, and it
|
|
73
|
+
# is what the error has to be passed on to avoid: a ramp becomes two flat
|
|
74
|
+
# bands with a step where it crosses a half.
|
|
75
|
+
show(image.gt(0.5).int8, "rounded, each pixel on its own")
|
|
76
|
+
show(dither(image), "dithered, the error passed on")
|
|
77
|
+
|
|
78
|
+
# The same walk in Ruby, at a size worth timing.
|
|
79
|
+
def dither_in_ruby (image)
|
|
80
|
+
rows, columns = image.dim
|
|
81
|
+
out = Array.new(rows) { Array.new(columns, 0) }
|
|
82
|
+
work = Array.new(rows + 1) { Array.new(columns + 2, 0.0) }
|
|
83
|
+
rows.times { |y| columns.times { |x| work[y][x + 1] = image[y, x] } }
|
|
84
|
+
rows.times do |y|
|
|
85
|
+
columns.times do |x|
|
|
86
|
+
old = work[y][x + 1]
|
|
87
|
+
new = old > 0.5 ? 1.0 : 0.0
|
|
88
|
+
out[y][x] = new.to_i
|
|
89
|
+
error = old - new
|
|
90
|
+
work[y][x + 2] += error * 7.0 / 16.0
|
|
91
|
+
work[y + 1][x] += error * 3.0 / 16.0
|
|
92
|
+
work[y + 1][x + 1] += error * 5.0 / 16.0
|
|
93
|
+
work[y + 1][x + 2] += error * 1.0 / 16.0
|
|
94
|
+
end
|
|
95
|
+
end
|
|
96
|
+
out
|
|
97
|
+
end
|
|
98
|
+
|
|
99
|
+
large = picture(600, 600)
|
|
100
|
+
here = dither(large)
|
|
101
|
+
there = dither_in_ruby(large)
|
|
102
|
+
puts format("600x600, agrees with Ruby %s", here.to_a == there)
|
|
103
|
+
|
|
104
|
+
started = Process.clock_gettime(Process::CLOCK_MONOTONIC)
|
|
105
|
+
5.times { dither(large) }
|
|
106
|
+
compiled = (Process.clock_gettime(Process::CLOCK_MONOTONIC) - started) / 5
|
|
107
|
+
|
|
108
|
+
started = Process.clock_gettime(Process::CLOCK_MONOTONIC)
|
|
109
|
+
dither_in_ruby(large)
|
|
110
|
+
interpreted = Process.clock_gettime(Process::CLOCK_MONOTONIC) - started
|
|
111
|
+
|
|
112
|
+
puts format(" %.1f ms compiled, %.0f ms in Ruby (%.0fx)",
|
|
113
|
+
compiled * 1e3, interpreted * 1e3, interpreted / compiled)
|
|
114
|
+
|
|
115
|
+
# Right to left, with the offsets mirrored: the same rule, the same picture
|
|
116
|
+
# going in, and a different picture coming out. Nothing here is wrong with
|
|
117
|
+
# either -- the walk is part of what the algorithm says, which is why this is
|
|
118
|
+
# not a pass an expression over arrays could have been rearranged into.
|
|
119
|
+
def dither_backwards (image)
|
|
120
|
+
rows, columns = image.dim
|
|
121
|
+
out = CArray.int8(rows, columns)
|
|
122
|
+
work = CArray.double(rows + 1, columns + 2)
|
|
123
|
+
work[0..-2, 1..-2] = image
|
|
124
|
+
CArray.jit_for(1) { |z|
|
|
125
|
+
(0...rows).each { |y|
|
|
126
|
+
(columns - 1).step(0, -1) { |x|
|
|
127
|
+
old = work[y, x + 1]
|
|
128
|
+
new = old > 0.5 ? 1.0 : 0.0
|
|
129
|
+
out[y, x] = new
|
|
130
|
+
error = old - new
|
|
131
|
+
work[y, x] += error * 7.0 / 16.0
|
|
132
|
+
work[y + 1, x + 2] += error * 3.0 / 16.0
|
|
133
|
+
work[y + 1, x + 1] += error * 5.0 / 16.0
|
|
134
|
+
work[y + 1, x] += error * 1.0 / 16.0
|
|
135
|
+
}
|
|
136
|
+
}
|
|
137
|
+
}
|
|
138
|
+
out
|
|
139
|
+
end
|
|
140
|
+
|
|
141
|
+
differ = dither(large).ne(dither_backwards(large)).count(1)
|
|
142
|
+
puts
|
|
143
|
+
puts format("walked the other way, %d of %d pixels land differently (%.1f%%)",
|
|
144
|
+
differ, large.elements, 100.0 * differ / large.elements)
|
|
@@ -0,0 +1,115 @@
|
|
|
1
|
+
# Aggregating by a label, when the aggregate is not a sum.
|
|
2
|
+
#
|
|
3
|
+
# Two million readings, each tagged with the station it came from, and the
|
|
4
|
+
# question is what each station did. Counting them and adding them up are
|
|
5
|
+
# what `CArray#bincount` is for and it is very good at it -- a pass over the
|
|
6
|
+
# labels and a pass over the weights, in C, and nothing here beats that.
|
|
7
|
+
#
|
|
8
|
+
# The rest is the problem. A maximum per station, the reading where that
|
|
9
|
+
# maximum happened, how many readings passed a threshold: none of those is a
|
|
10
|
+
# sum, and an array expression has no way to say "add this cell to the slot
|
|
11
|
+
# its label names, and only if it is larger than what is there". So the
|
|
12
|
+
# whole-array answer is a pass per label -- select the label's cells, reduce
|
|
13
|
+
# them, repeat -- and the cost is the number of labels times the size of the
|
|
14
|
+
# data, no matter how few cells each label owns.
|
|
15
|
+
#
|
|
16
|
+
# A kernel says it the way it is meant: one walk, and each cell updates the
|
|
17
|
+
# slot its label names.
|
|
18
|
+
#
|
|
19
|
+
# ruby examples/applications/group_stats.rb
|
|
20
|
+
|
|
21
|
+
$LOAD_PATH.unshift(File.expand_path("../../lib", __dir__))
|
|
22
|
+
require "carray/jit"
|
|
23
|
+
|
|
24
|
+
READINGS = 2_000_000
|
|
25
|
+
STATIONS = 64
|
|
26
|
+
|
|
27
|
+
random = CArray::Rng.new(seed: 20260913)
|
|
28
|
+
|
|
29
|
+
# Stations of very different sizes: the square of a uniform draw lands most
|
|
30
|
+
# of the readings on the low-numbered ones.
|
|
31
|
+
station = CArray.double(READINGS)
|
|
32
|
+
station.random!(rng: random)
|
|
33
|
+
station = ((station ** 2) * STATIONS).int32
|
|
34
|
+
|
|
35
|
+
value = CArray.double(READINGS)
|
|
36
|
+
value.random!(rng: random)
|
|
37
|
+
value = value * 40.0
|
|
38
|
+
|
|
39
|
+
LIMIT = 35.0
|
|
40
|
+
|
|
41
|
+
count = CArray.int64(STATIONS)
|
|
42
|
+
total = CArray.double(STATIONS)
|
|
43
|
+
peak = CArray.double(STATIONS).fill(-Float::INFINITY)
|
|
44
|
+
peak_at = CArray.int64(STATIONS).fill(-1)
|
|
45
|
+
exceed = CArray.int64(STATIONS)
|
|
46
|
+
|
|
47
|
+
def summarise (station, value, count, total, peak, peak_at, exceed)
|
|
48
|
+
count.fill(0)
|
|
49
|
+
total.fill(0.0)
|
|
50
|
+
peak.fill(-Float::INFINITY)
|
|
51
|
+
peak_at.fill(-1)
|
|
52
|
+
exceed.fill(0)
|
|
53
|
+
CArray.jit_for(station.elements) { |i|
|
|
54
|
+
k = station[i]
|
|
55
|
+
v = value[i]
|
|
56
|
+
count[k] += 1
|
|
57
|
+
total[k] += v
|
|
58
|
+
exceed[k] += 1 if v > LIMIT
|
|
59
|
+
if v > peak[k]
|
|
60
|
+
peak[k] = v
|
|
61
|
+
peak_at[k] = i
|
|
62
|
+
end
|
|
63
|
+
}
|
|
64
|
+
end
|
|
65
|
+
|
|
66
|
+
summarise(station, value, count, total, peak, peak_at, exceed)
|
|
67
|
+
|
|
68
|
+
puts format("%d readings over %d stations", READINGS, STATIONS)
|
|
69
|
+
puts " station count mean peak at reading over #{LIMIT.to_i}"
|
|
70
|
+
[0, 1, 2, STATIONS / 2, STATIONS - 1].each do |k|
|
|
71
|
+
puts format(" %7d %7d %9.3f %8.3f %12d %8d",
|
|
72
|
+
k, count[k], total[k] / count[k], peak[k], peak_at[k], exceed[k])
|
|
73
|
+
end
|
|
74
|
+
|
|
75
|
+
# Every one of those is checkable by selecting the station's cells, which is
|
|
76
|
+
# also the whole-array way of computing it in the first place.
|
|
77
|
+
k = 1
|
|
78
|
+
cells = value[station.eq(k)]
|
|
79
|
+
puts
|
|
80
|
+
puts format(" station %d checks out %s", k,
|
|
81
|
+
[count[k], peak[k], exceed[k]] ==
|
|
82
|
+
[cells.elements, cells.max, cells.gt(LIMIT).count(1)])
|
|
83
|
+
|
|
84
|
+
# What the sum and the count cost when CArray does them, which is the part of
|
|
85
|
+
# this a kernel has no business replacing.
|
|
86
|
+
def timed (repeats = 5)
|
|
87
|
+
started = Process.clock_gettime(Process::CLOCK_MONOTONIC)
|
|
88
|
+
repeats.times { yield }
|
|
89
|
+
(Process.clock_gettime(Process::CLOCK_MONOTONIC) - started) / repeats
|
|
90
|
+
end
|
|
91
|
+
|
|
92
|
+
binned = timed { station.bincount(length: STATIONS)
|
|
93
|
+
station.bincount(weights: value, length: STATIONS) }
|
|
94
|
+
|
|
95
|
+
# The maximum per station, over whole arrays: one selection and one reduction
|
|
96
|
+
# per station. The readings are walked STATIONS times over.
|
|
97
|
+
by_masks = timed(1) {
|
|
98
|
+
STATIONS.times.map { |label| value[station.eq(label)].max }
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
everything = timed {
|
|
102
|
+
summarise(station, value, count, total, peak, peak_at, exceed)
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
puts
|
|
106
|
+
puts format(" count and sum, CArray#bincount %6.1f ms", binned * 1e3)
|
|
107
|
+
puts format(" the maximum alone, by selection %6.1f ms", by_masks * 1e3)
|
|
108
|
+
puts format(" all five in one walk, this kernel %6.1f ms %.0fx",
|
|
109
|
+
everything * 1e3, by_masks / everything)
|
|
110
|
+
|
|
111
|
+
# Which is the shape of it: bincount walks the data twice whatever the labels
|
|
112
|
+
# are, the selections walk it once per label, and the kernel walks it once.
|
|
113
|
+
puts
|
|
114
|
+
puts format(" readings walked -- bincount %d, selections %d, kernel %d",
|
|
115
|
+
READINGS * 2, READINGS * STATIONS, READINGS)
|