carray-jit 0.1.2 → 0.1.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +771 -3
  3. data/README.md +7 -6
  4. data/carray-jit.gemspec +1 -3
  5. data/docs/00_Introduction.md +4 -3
  6. data/docs/01_GettingStarted.md +1 -1
  7. data/docs/02_KernelShapes.md +93 -14
  8. data/docs/03_SupportedFeatures.md +582 -26
  9. data/docs/04_Compiling.md +33 -6
  10. data/docs/05_DesignNotes.md +3 -3
  11. data/docs/06_Cheatsheet.md +198 -5
  12. data/docs/07_StepByStep.ja.md +534 -0
  13. data/docs/07_StepByStep.md +535 -0
  14. data/examples/README.md +12 -0
  15. data/examples/applications/alarm.rb +121 -0
  16. data/examples/applications/collatz.rb +105 -0
  17. data/examples/applications/cubic_spline.rb +331 -0
  18. data/examples/applications/dithering.rb +144 -0
  19. data/examples/applications/group_stats.rb +115 -0
  20. data/examples/applications/lookup.rb +126 -0
  21. data/examples/applications/median_filter.rb +153 -0
  22. data/examples/applications/parcel_ascent.rb +220 -0
  23. data/examples/applications/point_in_polygon.rb +111 -0
  24. data/examples/applications/random_walk.rb +98 -0
  25. data/examples/applications/van_der_pol.rb +186 -0
  26. data/examples/applications/wet_bulb.rb +140 -0
  27. data/examples/features/10_complex.rb +14 -4
  28. data/examples/features/15_loops.rb +7 -1
  29. data/lib/carray/jit/access.rb +14 -0
  30. data/lib/carray/jit/analyzer.rb +2077 -136
  31. data/lib/carray/jit/block_reader.rb +37 -6
  32. data/lib/carray/jit/c_function.rb +613 -76
  33. data/lib/carray/jit/c_generator.rb +1595 -156
  34. data/lib/carray/jit/call.rb +68 -0
  35. data/lib/carray/jit/compiler.rb +75 -11
  36. data/lib/carray/jit/kernel.rb +369 -32
  37. data/lib/carray/jit/node.rb +359 -9
  38. data/lib/carray/jit/sorting_networks.rb +182 -0
  39. data/lib/carray/jit/type_assignment.rb +371 -34
  40. data/lib/carray/jit/version.rb +1 -1
  41. data/lib/carray/jit.rb +560 -64
  42. metadata +22 -8
  43. data/ext/carray_jit_access/carray_jit_access.c +0 -460
  44. data/ext/carray_jit_access/extconf.rb +0 -8
@@ -0,0 +1,105 @@
1
+ # Collatz: how long each number takes to fall to 1.
2
+ #
3
+ # Halve it when it is even, take 3n+1 when it is odd, and count the steps
4
+ # until it reaches 1. How many steps that is depends on the number, and
5
+ # nothing about the number says in advance -- 3 takes 7 steps and 27 takes
6
+ # 111 -- so the length of the loop is the cell's own business and is what
7
+ # `while` is for.
8
+ #
9
+ # Written over whole arrays this is a pass per step over everything still
10
+ # running, and every cell pays for the longest one; there is no expression
11
+ # that lets one cell stop. Here each cell stops when it reaches 1, and the
12
+ # two answers -- the step count and the highest value on the way -- are
13
+ # written in the same pass.
14
+ #
15
+ # ruby examples/applications/collatz.rb
16
+
17
+ $LOAD_PATH.unshift(File.expand_path("../../lib", __dir__))
18
+ require "carray/jit"
19
+
20
+ N = 1_000_000
21
+
22
+ def collatz (n)
23
+ steps = CArray.int32(n)
24
+ peak = CArray.int64(n)
25
+ kernel = CArray.jit_for(1...n) { |i|
26
+ x = i
27
+ hi = x
28
+ count = 0
29
+ while x != 1
30
+ x = x % 2 == 0 ? x / 2 : 3 * x + 1
31
+ hi = x if x > hi
32
+ count = count + 1
33
+ end
34
+ steps[i] = count
35
+ peak[i] = hi
36
+ }
37
+ [steps, peak, kernel]
38
+ end
39
+
40
+ steps, peak, kernel = collatz(N)
41
+
42
+ # The numbers that took longer than anything before them.
43
+ puts "record holders below #{N}"
44
+ record = 0
45
+ (1...N).each do |i|
46
+ next unless steps[i] > record
47
+ record = steps[i]
48
+ next unless record > 400
49
+ puts format(" %7d %3d steps, reaching %d", i, record, peak[i])
50
+ end
51
+ puts format(" the highest value reached by any of them is %d, from %d",
52
+ peak.max, peak.max_addr)
53
+
54
+ # `x` is an int64 in the generated C, and nothing here asked for that. A
55
+ # local's type is settled from the whole body: `hi` is what `peak` is given,
56
+ # `peak` is an int64 array, and `hi = x` carries that back to `x`. Which is
57
+ # why 704511 reaches 56 billion without overflowing an int32 on the way.
58
+ puts
59
+ puts "what the local was compiled as"
60
+ puts format(" %s", kernel.c_source.lines.grep(/^\s*int\d+_t x;/).first.strip)
61
+
62
+ # The same walk in Ruby, for the answer and the time. Over the first tenth,
63
+ # because the point of the comparison is the ratio.
64
+ CHECK = N / 10
65
+
66
+ def collatz_in_ruby (n)
67
+ steps = Array.new(n, 0)
68
+ (1...n).each do |i|
69
+ x = i
70
+ count = 0
71
+ while x != 1
72
+ x = x.even? ? x / 2 : 3 * x + 1
73
+ count += 1
74
+ end
75
+ steps[i] = count
76
+ end
77
+ steps
78
+ end
79
+
80
+ in_ruby = collatz_in_ruby(CHECK)
81
+ puts
82
+ puts format("agrees with Ruby over 1...%d %s",
83
+ CHECK, steps[0...CHECK].to_a == in_ruby)
84
+
85
+ collatz(CHECK) # compiled and cached on the first call
86
+
87
+ started = Process.clock_gettime(Process::CLOCK_MONOTONIC)
88
+ 5.times { collatz(CHECK) }
89
+ compiled = (Process.clock_gettime(Process::CLOCK_MONOTONIC) - started) / 5
90
+
91
+ started = Process.clock_gettime(Process::CLOCK_MONOTONIC)
92
+ collatz_in_ruby(CHECK)
93
+ interpreted = Process.clock_gettime(Process::CLOCK_MONOTONIC) - started
94
+
95
+ puts format("%d starting values: %.0f ms compiled, %.0f ms in Ruby (%.0fx)",
96
+ CHECK, compiled * 1e3, interpreted * 1e3, interpreted / compiled)
97
+
98
+ # And the loop the kernel runs is the one the block states: `x.even?` is not
99
+ # in the subset, so the parity test is written as the arithmetic it is.
100
+ begin
101
+ CArray.jit_for(1) { |i| steps[i] = i.even? ? 0 : 1 }
102
+ rescue CArray::JIT::Unsupported => error
103
+ puts
104
+ puts "refused: #{error.message.sub(/ \(at line.*/m, "")}"
105
+ end
@@ -0,0 +1,331 @@
1
+ # A cubic spline through measured points, natural and clamped
2
+ #
3
+ # Fitting a C2 curve through n points is a tridiagonal solve: the second
4
+ # derivatives at the knots are the unknowns, and each interior knot gives one
5
+ # row with three entries in it. The rows are diagonally dominant, so the
6
+ # Thomas algorithm applies -- two sequential sweeps, which is the part an
7
+ # array library cannot do for you.
8
+ #
9
+ # The boundary conditions differ only in the first and last row. Natural
10
+ # asks for zero curvature at the ends, and needs nothing but the samples;
11
+ # clamped asks for a given slope there, and needs to be told what it is. What
12
+ # that buys is visible below: where the true curve is still bending at the end,
13
+ # the natural spline flattens it, and the error there is several times what it
14
+ # is anywhere inside -- the clamped fit removes it. Not-a-knot asks for the
15
+ # first two pieces to be one cubic, and the last two as well, which costs no
16
+ # information at all and recovers a good part of the difference.
17
+ #
18
+ # Evaluating the result is a second kernel, and a different shape of one --
19
+ # every query point searches for its interval on its own, so the body is a
20
+ # bisection with the bound in the extent. That search is where the time goes,
21
+ # and it is not always needed: query points that arrive sorted -- resampling
22
+ # onto a grid gives that -- let the interval be carried forward instead of
23
+ # looked up, which is a sweep of the kind the solver already is. Both are
24
+ # here, and they agree bit for bit.
25
+ #
26
+ # ruby examples/applications/cubic_spline.rb
27
+
28
+ $LOAD_PATH.unshift(File.expand_path("../../lib", __dir__))
29
+ require "carray/jit"
30
+
31
+ # The curve being sampled, and its slope -- known here so that the clamped
32
+ # ends have something to be given, and so the error can be measured.
33
+ def curve (x) Math.exp(-0.35 * x) * Math.sin(2.0 * x) end
34
+ def slope (x) Math.exp(-0.35 * x) * (2.0 * Math.cos(2.0 * x) - 0.35 * Math.sin(2.0 * x)) end
35
+
36
+ # ------------------------------------------------------------ the two kernels
37
+
38
+ # The moments M[i] = S''(x[i]). Interior rows come from the continuity of the
39
+ # first derivative; the two end rows are the boundary condition and are the
40
+ # only thing the three of them disagree about.
41
+ #
42
+ # Not-a-knot is the one that does not fit in a row of its own: asking S''' to
43
+ # be continuous at x[1] puts M[0], M[1] and M[2] in one equation, which is a
44
+ # band too wide. Stated the other way round it says M is linear across x[1],
45
+ # so M[0] is an extrapolation of the two inside it -- drop the first and last
46
+ # unknown, solve the n-2 rows between them, and put the two back afterwards.
47
+ # That is what `first_row` and `last_row` are: the same two sweeps, over the
48
+ # rows that are actually unknown.
49
+ def moments (x, y, n, work, ends)
50
+ a, b, c, d, cc, dd, moment = work
51
+
52
+ CArray.jit_for(1...(n-1)) { |i|
53
+ left = x[i] - x[i-1]
54
+ right = x[i+1] - x[i]
55
+ a[i] = left
56
+ b[i] = 2.0 * (left + right)
57
+ c[i] = right
58
+ d[i] = 6.0 * ((y[i+1] - y[i]) / right - (y[i] - y[i-1]) / left)
59
+ }
60
+
61
+ case ends
62
+ when nil # natural: S'' = 0 at both ends
63
+ first_row, last_row = 0, n-1
64
+ b[0] = 1.0 ; c[0] = 0.0 ; d[0] = 0.0
65
+ a[n-1] = 0.0 ; b[n-1] = 1.0 ; d[n-1] = 0.0
66
+ when :not_a_knot # not-a-knot: S''' continuous at x[1], x[n-2]
67
+ raise ArgumentError, "not-a-knot needs at least 4 points, got #{n}" if n < 4
68
+ first_row, last_row = 1, n-2
69
+ h0 = x[1] - x[0] ; h1 = x[2] - x[1] # d[1] and d[n-2] are the interior
70
+ b[1] = 3.0 * h0 + 2.0 * h1 + h0 * h0 / h1 # right-hand sides already
71
+ c[1] = (h1 * h1 - h0 * h0) / h1
72
+ hm = x[n-2] - x[n-3] ; hn = x[n-1] - x[n-2]
73
+ a[n-2] = (hm * hm - hn * hn) / hm
74
+ b[n-2] = 2.0 * hm + 3.0 * hn + hn * hn / hm
75
+ else # clamped: S' given at both ends
76
+ first_row, last_row = 0, n-1
77
+ first, last = ends
78
+ h0 = x[1] - x[0]
79
+ b[0] = 2.0 * h0 ; c[0] = h0
80
+ d[0] = 6.0 * ((y[1] - y[0]) / h0 - first)
81
+ hn = x[n-1] - x[n-2]
82
+ a[n-1] = hn ; b[n-1] = 2.0 * hn
83
+ d[n-1] = 6.0 * (last - (y[n-1] - y[n-2]) / hn)
84
+ end
85
+
86
+ cc[first_row] = c[first_row] / b[first_row] # Thomas, forward
87
+ dd[first_row] = d[first_row] / b[first_row]
88
+ CArray.jit_for((first_row+1)..last_row) { |i|
89
+ denominator = b[i] - a[i] * cc[i-1]
90
+ cc[i] = c[i] / denominator
91
+ dd[i] = (d[i] - a[i] * dd[i-1]) / denominator
92
+ }
93
+
94
+ moment[last_row] = dd[last_row] # Thomas, back substitution
95
+ CArray.jit_for((last_row-1).step(first_row, -1)) { |i|
96
+ moment[i] = dd[i] - cc[i] * moment[i+1]
97
+ }
98
+
99
+ if ends == :not_a_knot # the rows that were left out
100
+ moment[0] = moment[1] - (x[1] - x[0]) * (moment[2] - moment[1]) / (x[2] - x[1])
101
+ moment[n-1] = moment[n-2] + (x[n-1] - x[n-2]) * (moment[n-2] - moment[n-3]) / (x[n-2] - x[n-3])
102
+ end
103
+ moment
104
+ end
105
+
106
+ # Evaluate at arbitrary points. The knots are not equally spaced, so each
107
+ # query point has to find its interval; `passes` bounds the bisection, and a
108
+ # kernel that cannot fail to stop is the better kind.
109
+ def evaluate (x, y, moment, n, query, value, derivative)
110
+ points = query.dim[0]
111
+ passes = Math.log2(n).ceil + 1
112
+
113
+ CArray.jit_for(points) { |k|
114
+ lo = 0
115
+ hi = n - 2
116
+ (0...passes).each { |pass|
117
+ break if lo >= hi
118
+ middle = (lo + hi + 1) / 2
119
+ if x[middle] <= query[k]
120
+ lo = middle
121
+ else
122
+ hi = middle - 1
123
+ end
124
+ }
125
+ h = x[lo+1] - x[lo]
126
+ t = query[k] - x[lo]
127
+ linear = (y[lo+1] - y[lo]) / h - h * (2.0 * moment[lo] + moment[lo+1]) / 6.0
128
+ cubic = (moment[lo+1] - moment[lo]) / (6.0 * h)
129
+ value[k] = y[lo] + t * (linear + t * (0.5 * moment[lo] + t * cubic))
130
+ derivative[k] = linear + t * (moment[lo] + t * 3.0 * cubic)
131
+ }
132
+ end
133
+
134
+ # The same evaluation where the query points are sorted. Then nothing has to
135
+ # search: the interval only moves forward, so one sweep carries it from cell to
136
+ # cell and passes each knot once -- O(m + n) against O(m log n). How far it
137
+ # advances at one cell is the data's business, which is what `while` is for;
138
+ # there is no bound to put in an extent.
139
+ #
140
+ # The first cell has no k-1 to read, so it is seeded here and the sweep starts
141
+ # at 1 -- `jit_for(points)` over a body reading interval[k-1] is refused, and
142
+ # says so, exactly as the Thomas sweeps above are stated from 1.
143
+ #
144
+ # The intervals are found first and evaluated second, which is what keeps the
145
+ # seed from needing a copy of the polynomial: it is one entry in `interval`,
146
+ # and the second kernel does every point the same way. Fusing the two into a
147
+ # single kernel is worth about 30% more, at that price.
148
+ def resample (x, y, moment, n, query, value, derivative, interval)
149
+ points = query.dim[0]
150
+
151
+ seed = 0
152
+ seed += 1 while seed < n - 2 && x[seed+1] <= query[0]
153
+ interval[0] = seed
154
+
155
+ CArray.jit_for(1...points) { |k|
156
+ lo = interval[k-1]
157
+ while lo < n - 2 && x[lo+1] <= query[k]
158
+ lo = lo + 1
159
+ end
160
+ interval[k] = lo
161
+ }
162
+
163
+ CArray.jit_for(points) { |k|
164
+ lo = interval[k]
165
+ h = x[lo+1] - x[lo]
166
+ t = query[k] - x[lo]
167
+ linear = (y[lo+1] - y[lo]) / h - h * (2.0 * moment[lo] + moment[lo+1]) / 6.0
168
+ cubic = (moment[lo+1] - moment[lo]) / (6.0 * h)
169
+ value[k] = y[lo] + t * (linear + t * (0.5 * moment[lo] + t * cubic))
170
+ derivative[k] = linear + t * (moment[lo] + t * 3.0 * cubic)
171
+ }
172
+ end
173
+
174
+ # ------------------------------------------------------------------ the fit
175
+
176
+ n = 15
177
+ random = Random.new(20260909)
178
+ # Knots that are not equally spaced -- measurements rarely are, and nothing
179
+ # above assumed they would be.
180
+ x = CArray.double(n) { |i| 6.0 * i / (n - 1) }
181
+ (1...(n-1)).each { |i| x[i] += random.rand(-0.12..0.12) }
182
+ y = CArray.double(n) { |i| curve(x[i]) }
183
+
184
+ work = Array.new(7) { CArray.double(n) }
185
+ natural = moments(x, y, n, work, nil).copy
186
+ clamped = moments(x, y, n, work, [slope(x[0]), slope(x[n-1])]).copy
187
+ not_a_knot = moments(x, y, n, work, :not_a_knot).copy
188
+
189
+ points = 601
190
+ query = CArray.double(points) { |k| x[0] + (x[n-1] - x[0]) * k / (points - 1) }
191
+ truth = CArray.double(points) { |k| curve(query[k]) }
192
+
193
+ value = CArray.double(points)
194
+ derivative = CArray.double(points)
195
+
196
+ puts "cubic spline through #{n} unevenly spaced points, sampled at #{points}"
197
+
198
+ results = {}
199
+ { "natural" => natural, "clamped" => clamped, "not-a-knot" => not_a_knot }.each do |name, moment|
200
+ evaluate(x, y, moment, n, query, value, derivative)
201
+ results[name] = [value.copy, derivative.copy]
202
+
203
+ error = (value - truth).abs
204
+ edge = x[1] - x[0] # the first and last interval
205
+ ends = (0...points).select { |k| query[k] < x[0] + edge || query[k] > x[n-1] - edge }
206
+ inside = (0...points).to_a - ends
207
+
208
+ puts format(" %-10s max error %.2e overall, %.2e in the end intervals, %.2e inside",
209
+ name, error.max, ends.map { |k| error[k] }.max, inside.map { |k| error[k] }.max)
210
+ end
211
+
212
+ # What each boundary condition asked for, checked at the ends.
213
+ puts format(" natural S''(a) = %.1e, S''(b) = %.1e -- zero, by construction",
214
+ natural[0], natural[n-1])
215
+ puts format(" clamped S'(a) = %+.6f vs %+.6f asked for", results["clamped"][1][0], slope(x[0]))
216
+ puts format(" S'(b) = %+.6f vs %+.6f", results["clamped"][1][points-1], slope(x[n-1]))
217
+ jump = lambda { |m, i, j, k| # S''' across the knot that is not one
218
+ (m[j] - m[i]) / (x[j] - x[i]) - (m[k] - m[j]) / (x[k] - x[j])
219
+ }
220
+ puts format(" not-knot S''' jumps by %.1e at x[1] and %.1e at x[n-2] -- neither is a knot",
221
+ jump.call(not_a_knot, 0, 1, 2), jump.call(not_a_knot, n-3, n-2, n-1))
222
+
223
+ # The interpolation itself: all of them pass through every knot, so all three
224
+ # are asked, not just the one whose name comes first.
225
+ knots = CArray.double(n) { |i| x[i] }
226
+ at_knots = CArray.double(n)
227
+ at_knots_slope = CArray.double(n)
228
+ { "natural" => natural, "clamped" => clamped, "not-a-knot" => not_a_knot }.each do |name, moment|
229
+ evaluate(x, y, moment, n, knots, at_knots, at_knots_slope)
230
+ puts format(" %-10s max |S(x_i) - y_i| = %.2e", name, (at_knots - y).abs.max)
231
+ end
232
+
233
+ # The query grid above is sorted, so the sweep applies to it -- and gives back
234
+ # the same doubles, not merely close ones: the interval a point lands in is the
235
+ # same interval, and the polynomial evaluated in it is the same expression.
236
+ swept = CArray.double(points)
237
+ swept_slope = CArray.double(points)
238
+ resample(x, y, clamped, n, query, swept, swept_slope, CArray.int32(points))
239
+ puts " the sorted sweep agrees bit for bit #{swept.to_a == results["clamped"][0].to_a}"
240
+
241
+ # The curve, and the samples it was built from.
242
+ rows, columns = 15, 74
243
+ low, high = -0.55, 0.85
244
+ canvas = Array.new(rows) { " " * columns }
245
+ plot = lambda { |xs, ys, mark|
246
+ xs.each_with_index do |xv, k|
247
+ column = ((xv - x[0]) / (x[n-1] - x[0]) * (columns - 1)).round
248
+ row = ((high - ys[k]) / (high - low) * (rows - 1)).round
249
+ canvas[row][column] = mark if row.between?(0, rows - 1) && column.between?(0, columns - 1)
250
+ end
251
+ }
252
+ plot.call(query.to_a, results["clamped"][0].to_a, ".")
253
+ plot.call(x.to_a, y.to_a, "o")
254
+ puts
255
+ canvas.each { |row| puts " |#{row}|" }
256
+ puts " o samples, . the clamped spline through them"
257
+
258
+ # ------------------------------------------------------------------- the cost
259
+
260
+ def time (repeats)
261
+ started = Process.clock_gettime(Process::CLOCK_MONOTONIC)
262
+ repeats.times { yield }
263
+ (Process.clock_gettime(Process::CLOCK_MONOTONIC) - started) / repeats
264
+ end
265
+
266
+ # The same two kernels written as Ruby loops. The fit is O(n) and the
267
+ # evaluation O(m log n), so both grow slowly -- what the compiled version
268
+ # removes is the per-element cost, which is where all of the difference is.
269
+ def ruby_moments (x, y, n, work)
270
+ a, b, c, d, cc, dd, moment = work
271
+ (1...(n-1)).each do |i|
272
+ left = x[i] - x[i-1]
273
+ right = x[i+1] - x[i]
274
+ a[i] = left
275
+ b[i] = 2.0 * (left + right)
276
+ c[i] = right
277
+ d[i] = 6.0 * ((y[i+1] - y[i]) / right - (y[i] - y[i-1]) / left)
278
+ end
279
+ b[0] = 1.0 ; c[0] = 0.0 ; d[0] = 0.0
280
+ a[n-1] = 0.0 ; b[n-1] = 1.0 ; d[n-1] = 0.0
281
+ cc[0] = c[0] / b[0]
282
+ dd[0] = d[0] / b[0]
283
+ (1...n).each do |i|
284
+ denominator = b[i] - a[i] * cc[i-1]
285
+ cc[i] = c[i] / denominator
286
+ dd[i] = (d[i] - a[i] * dd[i-1]) / denominator
287
+ end
288
+ moment[n-1] = dd[n-1]
289
+ (n-2).step(0, -1) do |i|
290
+ moment[i] = dd[i] - cc[i] * moment[i+1]
291
+ end
292
+ moment
293
+ end
294
+
295
+ def ruby_evaluate (x, y, moment, n, query, value, derivative)
296
+ query.dim[0].times do |k|
297
+ lo, hi = 0, n - 2
298
+ while lo < hi
299
+ middle = (lo + hi + 1) / 2
300
+ if x[middle] <= query[k] then lo = middle else hi = middle - 1 end
301
+ end
302
+ h = x[lo+1] - x[lo]
303
+ t = query[k] - x[lo]
304
+ linear = (y[lo+1] - y[lo]) / h - h * (2.0 * moment[lo] + moment[lo+1]) / 6.0
305
+ cubic = (moment[lo+1] - moment[lo]) / (6.0 * h)
306
+ value[k] = y[lo] + t * (linear + t * (0.5 * moment[lo] + t * cubic))
307
+ derivative[k] = linear + t * (moment[lo] + t * 3.0 * cubic)
308
+ end
309
+ end
310
+
311
+ puts
312
+ [[200, 20_000], [2_000, 200_000]].each do |size, sampled|
313
+ knots = CArray.double(size) { |i| 6.0 * i / (size - 1) }
314
+ values = CArray.double(size) { |i| curve(knots[i]) }
315
+ scratch = Array.new(7) { CArray.double(size) }
316
+ at = CArray.double(sampled) { |k| 6.0 * k / (sampled - 1) }
317
+ out, slopes = CArray.double(sampled), CArray.double(sampled)
318
+
319
+ fit = time(20) { moments(knots, values, size, scratch, nil) }
320
+ fit_ruby = time(3) { ruby_moments(knots, values, size, scratch) }
321
+ moment = moments(knots, values, size, scratch, nil)
322
+ sample = time(5) { evaluate(knots, values, moment, size, at, out, slopes) }
323
+ sample_ruby = time(2) { ruby_evaluate(knots, values, moment, size, at, out, slopes) }
324
+ cells = CArray.int32(sampled)
325
+ swept = time(5) { resample(knots, values, moment, size, at, out, slopes, cells) }
326
+
327
+ puts format(" n = %5d fit %7.1f us vs %8.1f us Ruby (%3.0fx)", size, fit * 1e6, fit_ruby * 1e6, fit_ruby / fit)
328
+ puts format(" m = %6d eval %7.1f us vs %8.1f us Ruby (%3.0fx)", sampled, sample * 1e6, sample_ruby * 1e6, sample_ruby / sample)
329
+ puts format(" sorted %7.1f us -- the same points, with the search taken out (%.1fx)",
330
+ swept * 1e6, sample / swept)
331
+ end
@@ -0,0 +1,144 @@
1
+ # Floyd-Steinberg dithering: one bit per pixel, and the error passed on.
2
+ #
3
+ # Each pixel is rounded to black or white, and what the rounding threw away is
4
+ # handed to the neighbours that have not been visited yet -- seven sixteenths
5
+ # to the right, and the rest to the row below. So a cell writes the cells the
6
+ # loop is about to read, and the order it visits them in is not an
7
+ # optimisation but the definition: run the same rule right to left and a
8
+ # different picture comes out.
9
+ #
10
+ # That is what separates this from `sobel_edges.rb`, where every cell only
11
+ # reads its neighbours and the pass is a stencil. Here there is no window, no
12
+ # expression over whole arrays, and no order to be derived -- there is a walk,
13
+ # and the walk is the algorithm. The kernel is one cell whose body is that
14
+ # walk, which is the shape to reach for when the order is the point.
15
+ #
16
+ # ruby examples/applications/dithering.rb
17
+
18
+ $LOAD_PATH.unshift(File.expand_path("../../lib", __dir__))
19
+ require "carray/jit"
20
+
21
+ # A grey ramp, top to bottom, with a brighter disc sitting in it.
22
+ def picture (rows, columns)
23
+ image = CArray.double(rows, columns)
24
+ CArray.jit_for(rows, columns) { |y, x|
25
+ dy = (y - rows / 2.0) / (rows / 2.0)
26
+ dx = (x - columns / 2.0) / (columns / 2.0) * 0.42
27
+ ramp = 0.08 + 0.84 * y / rows
28
+ image[y, x] = dx * dx + dy * dy < 0.16 ? ramp * 0.35 + 0.62 : ramp
29
+ }
30
+ image
31
+ end
32
+
33
+ # The work array carries a row below and a column on each side, so that the
34
+ # four neighbours a pixel writes are always cells that exist. A kernel's
35
+ # bounds are checked against the subscripts it is written with rather than
36
+ # against the branches that guard them, so the room is made in the array, not
37
+ # in an `if`.
38
+ def dither (image)
39
+ rows, columns = image.dim
40
+ out = CArray.int8(rows, columns)
41
+ work = CArray.double(rows + 1, columns + 2)
42
+ work[0..-2, 1..-2] = image
43
+ CArray.jit_for(1) { |z|
44
+ (0...rows).each { |y|
45
+ (0...columns).each { |x|
46
+ old = work[y, x + 1]
47
+ new = old > 0.5 ? 1.0 : 0.0
48
+ out[y, x] = new
49
+ error = old - new
50
+ work[y, x + 2] += error * 7.0 / 16.0
51
+ work[y + 1, x] += error * 3.0 / 16.0
52
+ work[y + 1, x + 1] += error * 5.0 / 16.0
53
+ work[y + 1, x + 2] += error * 1.0 / 16.0
54
+ }
55
+ }
56
+ }
57
+ out
58
+ end
59
+
60
+ ROWS, COLUMNS = 22, 72
61
+
62
+ def show (bits, title)
63
+ puts title
64
+ bits.dim[0].times do |y|
65
+ puts " " + (0...bits.dim[1]).map { |x| bits[y, x] == 1 ? "@" : " " }.join
66
+ end
67
+ puts
68
+ end
69
+
70
+ image = picture(ROWS, COLUMNS)
71
+
72
+ # Rounding each pixel on its own is an expression over the whole array, and it
73
+ # is what the error has to be passed on to avoid: a ramp becomes two flat
74
+ # bands with a step where it crosses a half.
75
+ show(image.gt(0.5).int8, "rounded, each pixel on its own")
76
+ show(dither(image), "dithered, the error passed on")
77
+
78
+ # The same walk in Ruby, at a size worth timing.
79
+ def dither_in_ruby (image)
80
+ rows, columns = image.dim
81
+ out = Array.new(rows) { Array.new(columns, 0) }
82
+ work = Array.new(rows + 1) { Array.new(columns + 2, 0.0) }
83
+ rows.times { |y| columns.times { |x| work[y][x + 1] = image[y, x] } }
84
+ rows.times do |y|
85
+ columns.times do |x|
86
+ old = work[y][x + 1]
87
+ new = old > 0.5 ? 1.0 : 0.0
88
+ out[y][x] = new.to_i
89
+ error = old - new
90
+ work[y][x + 2] += error * 7.0 / 16.0
91
+ work[y + 1][x] += error * 3.0 / 16.0
92
+ work[y + 1][x + 1] += error * 5.0 / 16.0
93
+ work[y + 1][x + 2] += error * 1.0 / 16.0
94
+ end
95
+ end
96
+ out
97
+ end
98
+
99
+ large = picture(600, 600)
100
+ here = dither(large)
101
+ there = dither_in_ruby(large)
102
+ puts format("600x600, agrees with Ruby %s", here.to_a == there)
103
+
104
+ started = Process.clock_gettime(Process::CLOCK_MONOTONIC)
105
+ 5.times { dither(large) }
106
+ compiled = (Process.clock_gettime(Process::CLOCK_MONOTONIC) - started) / 5
107
+
108
+ started = Process.clock_gettime(Process::CLOCK_MONOTONIC)
109
+ dither_in_ruby(large)
110
+ interpreted = Process.clock_gettime(Process::CLOCK_MONOTONIC) - started
111
+
112
+ puts format(" %.1f ms compiled, %.0f ms in Ruby (%.0fx)",
113
+ compiled * 1e3, interpreted * 1e3, interpreted / compiled)
114
+
115
+ # Right to left, with the offsets mirrored: the same rule, the same picture
116
+ # going in, and a different picture coming out. Nothing here is wrong with
117
+ # either -- the walk is part of what the algorithm says, which is why this is
118
+ # not a pass an expression over arrays could have been rearranged into.
119
+ def dither_backwards (image)
120
+ rows, columns = image.dim
121
+ out = CArray.int8(rows, columns)
122
+ work = CArray.double(rows + 1, columns + 2)
123
+ work[0..-2, 1..-2] = image
124
+ CArray.jit_for(1) { |z|
125
+ (0...rows).each { |y|
126
+ (columns - 1).step(0, -1) { |x|
127
+ old = work[y, x + 1]
128
+ new = old > 0.5 ? 1.0 : 0.0
129
+ out[y, x] = new
130
+ error = old - new
131
+ work[y, x] += error * 7.0 / 16.0
132
+ work[y + 1, x + 2] += error * 3.0 / 16.0
133
+ work[y + 1, x + 1] += error * 5.0 / 16.0
134
+ work[y + 1, x] += error * 1.0 / 16.0
135
+ }
136
+ }
137
+ }
138
+ out
139
+ end
140
+
141
+ differ = dither(large).ne(dither_backwards(large)).count(1)
142
+ puts
143
+ puts format("walked the other way, %d of %d pixels land differently (%.1f%%)",
144
+ differ, large.elements, 100.0 * differ / large.elements)
@@ -0,0 +1,115 @@
1
+ # Aggregating by a label, when the aggregate is not a sum.
2
+ #
3
+ # Two million readings, each tagged with the station it came from, and the
4
+ # question is what each station did. Counting them and adding them up are
5
+ # what `CArray#bincount` is for and it is very good at it -- a pass over the
6
+ # labels and a pass over the weights, in C, and nothing here beats that.
7
+ #
8
+ # The rest is the problem. A maximum per station, the reading where that
9
+ # maximum happened, how many readings passed a threshold: none of those is a
10
+ # sum, and an array expression has no way to say "add this cell to the slot
11
+ # its label names, and only if it is larger than what is there". So the
12
+ # whole-array answer is a pass per label -- select the label's cells, reduce
13
+ # them, repeat -- and the cost is the number of labels times the size of the
14
+ # data, no matter how few cells each label owns.
15
+ #
16
+ # A kernel says it the way it is meant: one walk, and each cell updates the
17
+ # slot its label names.
18
+ #
19
+ # ruby examples/applications/group_stats.rb
20
+
21
+ $LOAD_PATH.unshift(File.expand_path("../../lib", __dir__))
22
+ require "carray/jit"
23
+
24
+ READINGS = 2_000_000
25
+ STATIONS = 64
26
+
27
+ random = CArray::Rng.new(seed: 20260913)
28
+
29
+ # Stations of very different sizes: the square of a uniform draw lands most
30
+ # of the readings on the low-numbered ones.
31
+ station = CArray.double(READINGS)
32
+ station.random!(rng: random)
33
+ station = ((station ** 2) * STATIONS).int32
34
+
35
+ value = CArray.double(READINGS)
36
+ value.random!(rng: random)
37
+ value = value * 40.0
38
+
39
+ LIMIT = 35.0
40
+
41
+ count = CArray.int64(STATIONS)
42
+ total = CArray.double(STATIONS)
43
+ peak = CArray.double(STATIONS).fill(-Float::INFINITY)
44
+ peak_at = CArray.int64(STATIONS).fill(-1)
45
+ exceed = CArray.int64(STATIONS)
46
+
47
+ def summarise (station, value, count, total, peak, peak_at, exceed)
48
+ count.fill(0)
49
+ total.fill(0.0)
50
+ peak.fill(-Float::INFINITY)
51
+ peak_at.fill(-1)
52
+ exceed.fill(0)
53
+ CArray.jit_for(station.elements) { |i|
54
+ k = station[i]
55
+ v = value[i]
56
+ count[k] += 1
57
+ total[k] += v
58
+ exceed[k] += 1 if v > LIMIT
59
+ if v > peak[k]
60
+ peak[k] = v
61
+ peak_at[k] = i
62
+ end
63
+ }
64
+ end
65
+
66
+ summarise(station, value, count, total, peak, peak_at, exceed)
67
+
68
+ puts format("%d readings over %d stations", READINGS, STATIONS)
69
+ puts " station count mean peak at reading over #{LIMIT.to_i}"
70
+ [0, 1, 2, STATIONS / 2, STATIONS - 1].each do |k|
71
+ puts format(" %7d %7d %9.3f %8.3f %12d %8d",
72
+ k, count[k], total[k] / count[k], peak[k], peak_at[k], exceed[k])
73
+ end
74
+
75
+ # Every one of those is checkable by selecting the station's cells, which is
76
+ # also the whole-array way of computing it in the first place.
77
+ k = 1
78
+ cells = value[station.eq(k)]
79
+ puts
80
+ puts format(" station %d checks out %s", k,
81
+ [count[k], peak[k], exceed[k]] ==
82
+ [cells.elements, cells.max, cells.gt(LIMIT).count(1)])
83
+
84
+ # What the sum and the count cost when CArray does them, which is the part of
85
+ # this a kernel has no business replacing.
86
+ def timed (repeats = 5)
87
+ started = Process.clock_gettime(Process::CLOCK_MONOTONIC)
88
+ repeats.times { yield }
89
+ (Process.clock_gettime(Process::CLOCK_MONOTONIC) - started) / repeats
90
+ end
91
+
92
+ binned = timed { station.bincount(length: STATIONS)
93
+ station.bincount(weights: value, length: STATIONS) }
94
+
95
+ # The maximum per station, over whole arrays: one selection and one reduction
96
+ # per station. The readings are walked STATIONS times over.
97
+ by_masks = timed(1) {
98
+ STATIONS.times.map { |label| value[station.eq(label)].max }
99
+ }
100
+
101
+ everything = timed {
102
+ summarise(station, value, count, total, peak, peak_at, exceed)
103
+ }
104
+
105
+ puts
106
+ puts format(" count and sum, CArray#bincount %6.1f ms", binned * 1e3)
107
+ puts format(" the maximum alone, by selection %6.1f ms", by_masks * 1e3)
108
+ puts format(" all five in one walk, this kernel %6.1f ms %.0fx",
109
+ everything * 1e3, by_masks / everything)
110
+
111
+ # Which is the shape of it: bincount walks the data twice whatever the labels
112
+ # are, the selections walk it once per label, and the kernel walks it once.
113
+ puts
114
+ puts format(" readings walked -- bincount %d, selections %d, kernel %d",
115
+ READINGS * 2, READINGS * STATIONS, READINGS)