carray-jit 0.1.2 → 0.1.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +771 -3
- data/README.md +7 -6
- data/carray-jit.gemspec +1 -3
- data/docs/00_Introduction.md +4 -3
- data/docs/01_GettingStarted.md +1 -1
- data/docs/02_KernelShapes.md +93 -14
- data/docs/03_SupportedFeatures.md +582 -26
- data/docs/04_Compiling.md +33 -6
- data/docs/05_DesignNotes.md +3 -3
- data/docs/06_Cheatsheet.md +198 -5
- data/docs/07_StepByStep.ja.md +534 -0
- data/docs/07_StepByStep.md +535 -0
- data/examples/README.md +12 -0
- data/examples/applications/alarm.rb +121 -0
- data/examples/applications/collatz.rb +105 -0
- data/examples/applications/cubic_spline.rb +331 -0
- data/examples/applications/dithering.rb +144 -0
- data/examples/applications/group_stats.rb +115 -0
- data/examples/applications/lookup.rb +126 -0
- data/examples/applications/median_filter.rb +153 -0
- data/examples/applications/parcel_ascent.rb +220 -0
- data/examples/applications/point_in_polygon.rb +111 -0
- data/examples/applications/random_walk.rb +98 -0
- data/examples/applications/van_der_pol.rb +186 -0
- data/examples/applications/wet_bulb.rb +140 -0
- data/examples/features/10_complex.rb +14 -4
- data/examples/features/15_loops.rb +7 -1
- data/lib/carray/jit/access.rb +14 -0
- data/lib/carray/jit/analyzer.rb +2077 -136
- data/lib/carray/jit/block_reader.rb +37 -6
- data/lib/carray/jit/c_function.rb +613 -76
- data/lib/carray/jit/c_generator.rb +1595 -156
- data/lib/carray/jit/call.rb +68 -0
- data/lib/carray/jit/compiler.rb +75 -11
- data/lib/carray/jit/kernel.rb +369 -32
- data/lib/carray/jit/node.rb +359 -9
- data/lib/carray/jit/sorting_networks.rb +182 -0
- data/lib/carray/jit/type_assignment.rb +371 -34
- data/lib/carray/jit/version.rb +1 -1
- data/lib/carray/jit.rb +560 -64
- metadata +22 -8
- data/ext/carray_jit_access/carray_jit_access.c +0 -460
- data/ext/carray_jit_access/extconf.rb +0 -8
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
# Half a million queries against an uneven grid.
|
|
2
|
+
#
|
|
3
|
+
# Tabulated data rarely comes on a regular grid -- a sounding is dense near
|
|
4
|
+
# the ground, a spectrum near a line -- so reading a value off it means
|
|
5
|
+
# finding which two samples a query falls between, and that is a search per
|
|
6
|
+
# query. The loop is short, about `log n` steps, but where it looks is
|
|
7
|
+
# different for every cell, which is the part an expression over whole arrays
|
|
8
|
+
# cannot express: it has no way to say "this cell reads knots[lo] where lo is
|
|
9
|
+
# what this cell just worked out".
|
|
10
|
+
#
|
|
11
|
+
# CArray can still do it, by materialising the answer to the search as an
|
|
12
|
+
# index array and gathering through it, and that is measured here beside the
|
|
13
|
+
# kernel. The two agree to the bit; what differs is how many passes and how
|
|
14
|
+
# much scratch it took -- and, in the search itself, what may be assumed
|
|
15
|
+
# about the grid.
|
|
16
|
+
#
|
|
17
|
+
# ruby examples/applications/lookup.rb
|
|
18
|
+
|
|
19
|
+
$LOAD_PATH.unshift(File.expand_path("../../lib", __dir__))
|
|
20
|
+
require "carray/jit"
|
|
21
|
+
|
|
22
|
+
N = 2_000 # samples in the table
|
|
23
|
+
Q = 500_000 # queries against it
|
|
24
|
+
|
|
25
|
+
# A grid that crowds towards zero, and a function sampled on it.
|
|
26
|
+
knots = (CArray.double(N).seq!(0, 1.0 / (N - 1)) ** 2) * 100
|
|
27
|
+
table = knots.sin
|
|
28
|
+
query = CArray.double(Q).seq!(0, 99.9 / Q)
|
|
29
|
+
|
|
30
|
+
# The search and the interpolation in one pass. `lo` and `hi` are the cell's
|
|
31
|
+
# own, so the kernel reads `knots[lo]` at an address no other cell shares --
|
|
32
|
+
# a gather, written as the subscript it is.
|
|
33
|
+
def interpolate (knots, table, query)
|
|
34
|
+
n = knots.elements
|
|
35
|
+
out = CArray.double(query.elements)
|
|
36
|
+
CArray.jit_for(query.elements) { |i|
|
|
37
|
+
t = query[i]
|
|
38
|
+
lo = 0
|
|
39
|
+
hi = n - 1
|
|
40
|
+
while hi - lo > 1
|
|
41
|
+
mid = (lo + hi) / 2
|
|
42
|
+
if knots[mid] > t
|
|
43
|
+
hi = mid
|
|
44
|
+
else
|
|
45
|
+
lo = mid
|
|
46
|
+
end
|
|
47
|
+
end
|
|
48
|
+
weight = (t - knots[lo]) / (knots[lo + 1] - knots[lo])
|
|
49
|
+
out[i] = table[lo] * (1.0 - weight) + table[lo + 1] * weight
|
|
50
|
+
}
|
|
51
|
+
out
|
|
52
|
+
end
|
|
53
|
+
|
|
54
|
+
# The same answer over whole arrays. `search_nearest` is CArray's own and is
|
|
55
|
+
# not a loop in Ruby -- but it answers with the nearest sample rather than the
|
|
56
|
+
# one below, so the bracket has to be fixed up, and every step from here on is
|
|
57
|
+
# another pass and another array the size of the queries.
|
|
58
|
+
#
|
|
59
|
+
# It is also the slowest line here, and that is a contract rather than a
|
|
60
|
+
# fault: nothing says a CArray is sorted, so `search_nearest` looks at every
|
|
61
|
+
# sample for every query. A kernel is where the grid being monotonic can be
|
|
62
|
+
# used, because the search is written rather than called.
|
|
63
|
+
def interpolate_over_arrays (knots, table, query)
|
|
64
|
+
near = knots.search_nearest(query)
|
|
65
|
+
lo = near.to_ca
|
|
66
|
+
lo[knots[near].gt(query)] -= 1
|
|
67
|
+
lo = lo.clip(0, knots.elements - 2)
|
|
68
|
+
hi = lo + 1
|
|
69
|
+
weight = (query - knots[lo]) / (knots[hi] - knots[lo])
|
|
70
|
+
table[lo] * (1.0 - weight) + table[hi] * weight
|
|
71
|
+
end
|
|
72
|
+
|
|
73
|
+
# And in Ruby, which is where this calculation usually lives.
|
|
74
|
+
def interpolate_in_ruby (knots, table, query)
|
|
75
|
+
ks = knots.to_a
|
|
76
|
+
ts = table.to_a
|
|
77
|
+
query.to_a.map { |t|
|
|
78
|
+
lo = 0
|
|
79
|
+
hi = ks.size - 1
|
|
80
|
+
while hi - lo > 1
|
|
81
|
+
mid = (lo + hi) / 2
|
|
82
|
+
if ks[mid] > t then hi = mid else lo = mid end
|
|
83
|
+
end
|
|
84
|
+
weight = (t - ks[lo]) / (ks[lo + 1] - ks[lo])
|
|
85
|
+
ts[lo] * (1.0 - weight) + ts[lo + 1] * weight
|
|
86
|
+
}
|
|
87
|
+
end
|
|
88
|
+
|
|
89
|
+
here = interpolate(knots, table, query)
|
|
90
|
+
there = interpolate_over_arrays(knots, table, query)
|
|
91
|
+
|
|
92
|
+
puts format("%d queries against %d uneven samples", Q, N)
|
|
93
|
+
puts format(" agrees with the whole-array route %s", (here - there).abs.max == 0.0)
|
|
94
|
+
puts format(" agrees with Ruby %s", here.to_a == interpolate_in_ruby(knots, table, query))
|
|
95
|
+
puts format(" worst interpolation error %.2e", (here - query.sin).abs.max)
|
|
96
|
+
puts format(" and where the grid is coarsest x = %.1f",
|
|
97
|
+
query[(here - query.sin).abs.max_addr])
|
|
98
|
+
|
|
99
|
+
def timed (repeats = 5)
|
|
100
|
+
started = Process.clock_gettime(Process::CLOCK_MONOTONIC)
|
|
101
|
+
repeats.times { yield }
|
|
102
|
+
(Process.clock_gettime(Process::CLOCK_MONOTONIC) - started) / repeats
|
|
103
|
+
end
|
|
104
|
+
|
|
105
|
+
compiled = timed { interpolate(knots, table, query) }
|
|
106
|
+
searched = timed { knots.search_nearest(query) }
|
|
107
|
+
whole = timed { interpolate_over_arrays(knots, table, query) }
|
|
108
|
+
interpreted = timed(1) { interpolate_in_ruby(knots, table, query) }
|
|
109
|
+
|
|
110
|
+
puts
|
|
111
|
+
puts format(" one kernel, search and all %7.1f ms", compiled * 1e3)
|
|
112
|
+
puts format(" whole arrays, after the search %7.1f ms %.1fx",
|
|
113
|
+
(whole - searched) * 1e3, (whole - searched) / compiled)
|
|
114
|
+
puts format(" whole arrays, with it %7.1f ms %.0fx",
|
|
115
|
+
whole * 1e3, whole / compiled)
|
|
116
|
+
puts format(" Ruby, bisecting per query %7.1f ms %.0fx",
|
|
117
|
+
interpreted * 1e3, interpreted / compiled)
|
|
118
|
+
|
|
119
|
+
# The middle line is the one about style: six passes over half a million
|
|
120
|
+
# elements, and the arrays to hold them, against one pass that keeps `lo` in a
|
|
121
|
+
# register. The line below it is about the search, and belongs to
|
|
122
|
+
# `search_nearest`'s contract rather than to arrays.
|
|
123
|
+
puts
|
|
124
|
+
puts format(" arrays of %d elements the whole-array route holds: 7", Q)
|
|
125
|
+
puts format(" the kernel holds none, and bisects in %d comparisons",
|
|
126
|
+
Math.log2(N).ceil)
|
|
@@ -0,0 +1,153 @@
|
|
|
1
|
+
# Despiking a series with a median, and where a cell keeps its scratch.
|
|
2
|
+
#
|
|
3
|
+
# A mean is ruined by one bad sample; a median is not, which is why a median
|
|
4
|
+
# filter is what a record with spikes in it gets. The window has to be
|
|
5
|
+
# ordered to find its middle, and ordering a window is the thing an array
|
|
6
|
+
# expression has no operation for: CArray can sort an array, but not the nine
|
|
7
|
+
# cells around every cell, separately, a million times.
|
|
8
|
+
#
|
|
9
|
+
# So this is three spellings of one answer -- Ruby's, a kernel that sorts the
|
|
10
|
+
# window in a workspace, and a kernel that finds the middle with comparisons
|
|
11
|
+
# alone -- and they agree to the bit. What separates them is where the cell's
|
|
12
|
+
# scratch lives. A workspace indexed by the cell is memory: a million rows of
|
|
13
|
+
# nine, written and read on every comparison. A local is a register. It is
|
|
14
|
+
# the same lesson every other kernel here quietly relies on, and this is where
|
|
15
|
+
# it is worth a hundred times.
|
|
16
|
+
#
|
|
17
|
+
# ruby examples/applications/median_filter.rb
|
|
18
|
+
|
|
19
|
+
$LOAD_PATH.unshift(File.expand_path("../../lib", __dir__))
|
|
20
|
+
require "carray/jit"
|
|
21
|
+
|
|
22
|
+
SAMPLES = 1_000_000
|
|
23
|
+
WINDOW = 9
|
|
24
|
+
HALF = WINDOW / 2
|
|
25
|
+
|
|
26
|
+
# A smooth signal, some noise, and a spike every few hundred samples.
|
|
27
|
+
random = CArray::Rng.new(seed: 20260913)
|
|
28
|
+
noise = CArray.double(SAMPLES)
|
|
29
|
+
noise.random!(rng: random)
|
|
30
|
+
chance = CArray.double(SAMPLES)
|
|
31
|
+
chance.random!(rng: random)
|
|
32
|
+
signal = CArray.double(SAMPLES)
|
|
33
|
+
clean = CArray.double(SAMPLES)
|
|
34
|
+
CArray.jit_for(SAMPLES) { |i|
|
|
35
|
+
smooth = 10.0 * Math.sin(i / 5000.0) + 0.5 * (noise[i] - 0.5)
|
|
36
|
+
clean[i] = smooth
|
|
37
|
+
signal[i] = chance[i] < 0.002 ? smooth + 40.0 * (chance[i] * 500.0 - 0.5) : smooth
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
sorted_out = CArray.double(SAMPLES)
|
|
41
|
+
network_out = CArray.double(SAMPLES)
|
|
42
|
+
|
|
43
|
+
# One: the window copied into the cell's own row of a workspace and put in
|
|
44
|
+
# order there. Straightforward, and every comparison is a memory access.
|
|
45
|
+
workspace = CArray.double(SAMPLES, WINDOW)
|
|
46
|
+
|
|
47
|
+
def by_sorting (signal, out, workspace)
|
|
48
|
+
n = signal.elements
|
|
49
|
+
CArray.jit_for(HALF...(n - HALF)) { |i|
|
|
50
|
+
k = 0
|
|
51
|
+
while k < WINDOW
|
|
52
|
+
workspace[i, k] = signal[i - HALF + k]
|
|
53
|
+
k = k + 1
|
|
54
|
+
end
|
|
55
|
+
k = 1
|
|
56
|
+
while k < WINDOW
|
|
57
|
+
value = workspace[i, k]
|
|
58
|
+
j = k - 1
|
|
59
|
+
while j >= 0 && workspace[i, j] > value
|
|
60
|
+
workspace[i, j + 1] = workspace[i, j]
|
|
61
|
+
j = j - 1
|
|
62
|
+
end
|
|
63
|
+
workspace[i, j + 1] = value
|
|
64
|
+
k = k + 1
|
|
65
|
+
end
|
|
66
|
+
out[i] = workspace[i, HALF]
|
|
67
|
+
}
|
|
68
|
+
end
|
|
69
|
+
|
|
70
|
+
# Two: the median of nine by comparisons alone. Order each group of three,
|
|
71
|
+
# then take the largest of the three smallest, the middle of the middles and
|
|
72
|
+
# the smallest of the largests, and the median of those three is the median
|
|
73
|
+
# of the nine. Nineteen comparisons, and nothing leaves a register.
|
|
74
|
+
def by_network (signal, out)
|
|
75
|
+
n = signal.elements
|
|
76
|
+
CArray.jit_for(HALF...(n - HALF)) { |i|
|
|
77
|
+
a = signal[i - 4]; b = signal[i - 3]; c = signal[i - 2]
|
|
78
|
+
d = signal[i - 1]; e = signal[i]; f = signal[i + 1]
|
|
79
|
+
g = signal[i + 2]; h = signal[i + 3]; k = signal[i + 4]
|
|
80
|
+
|
|
81
|
+
lo = a < b ? a : b; hi = a < b ? b : a
|
|
82
|
+
mid = hi < c ? hi : c
|
|
83
|
+
m1 = lo > mid ? lo : mid
|
|
84
|
+
l1 = lo < c ? lo : c
|
|
85
|
+
h1 = hi > c ? hi : c
|
|
86
|
+
|
|
87
|
+
lo = d < e ? d : e; hi = d < e ? e : d
|
|
88
|
+
mid = hi < f ? hi : f
|
|
89
|
+
m2 = lo > mid ? lo : mid
|
|
90
|
+
l2 = lo < f ? lo : f
|
|
91
|
+
h2 = hi > f ? hi : f
|
|
92
|
+
|
|
93
|
+
lo = g < h ? g : h; hi = g < h ? h : g
|
|
94
|
+
mid = hi < k ? hi : k
|
|
95
|
+
m3 = lo > mid ? lo : mid
|
|
96
|
+
l3 = lo < k ? lo : k
|
|
97
|
+
h3 = hi > k ? hi : k
|
|
98
|
+
|
|
99
|
+
largest_small = l1 > l2 ? l1 : l2
|
|
100
|
+
largest_small = largest_small > l3 ? largest_small : l3
|
|
101
|
+
smallest_large = h1 < h2 ? h1 : h2
|
|
102
|
+
smallest_large = smallest_large < h3 ? smallest_large : h3
|
|
103
|
+
|
|
104
|
+
lo = m1 < m2 ? m1 : m2; hi = m1 < m2 ? m2 : m1
|
|
105
|
+
mid = hi < m3 ? hi : m3
|
|
106
|
+
middle_middle = lo > mid ? lo : mid
|
|
107
|
+
|
|
108
|
+
lo = largest_small < middle_middle ? largest_small : middle_middle
|
|
109
|
+
hi = largest_small < middle_middle ? middle_middle : largest_small
|
|
110
|
+
mid = hi < smallest_large ? hi : smallest_large
|
|
111
|
+
out[i] = lo > mid ? lo : mid
|
|
112
|
+
}
|
|
113
|
+
end
|
|
114
|
+
|
|
115
|
+
by_sorting(signal, sorted_out, workspace)
|
|
116
|
+
by_network(signal, network_out)
|
|
117
|
+
|
|
118
|
+
# And Ruby's, which is what this would otherwise be.
|
|
119
|
+
def in_ruby (signal)
|
|
120
|
+
signal.to_a.each_cons(WINDOW).map { |window| window.sort[HALF] }
|
|
121
|
+
end
|
|
122
|
+
|
|
123
|
+
reference = in_ruby(signal)
|
|
124
|
+
inner = HALF...(SAMPLES - HALF)
|
|
125
|
+
puts format("%d samples, window of %d", SAMPLES, WINDOW)
|
|
126
|
+
puts format(" the sorted kernel agrees with Ruby %s",
|
|
127
|
+
inner.all? { |i| sorted_out[i] == reference[i - HALF] })
|
|
128
|
+
puts format(" the network kernel agrees with Ruby %s",
|
|
129
|
+
inner.all? { |i| network_out[i] == reference[i - HALF] })
|
|
130
|
+
|
|
131
|
+
puts format(" %d spikes went in; the worst departure from the clean signal was %.2f, and is now %.2f",
|
|
132
|
+
chance.lt(0.002).count(1),
|
|
133
|
+
(signal[inner] - clean[inner]).abs.max,
|
|
134
|
+
(network_out[inner] - clean[inner]).abs.max)
|
|
135
|
+
|
|
136
|
+
def timed (repeats = 3)
|
|
137
|
+
started = Process.clock_gettime(Process::CLOCK_MONOTONIC)
|
|
138
|
+
repeats.times { yield }
|
|
139
|
+
(Process.clock_gettime(Process::CLOCK_MONOTONIC) - started) / repeats
|
|
140
|
+
end
|
|
141
|
+
|
|
142
|
+
sorting = timed { by_sorting(signal, sorted_out, workspace) }
|
|
143
|
+
network = timed { by_network(signal, network_out) }
|
|
144
|
+
interpreted = timed(1) { in_ruby(signal) }
|
|
145
|
+
|
|
146
|
+
puts
|
|
147
|
+
puts format(" Ruby, sorting each window %7.0f ms", interpreted * 1e3)
|
|
148
|
+
puts format(" kernel, sorting in a workspace %7.0f ms %.0fx",
|
|
149
|
+
sorting * 1e3, interpreted / sorting)
|
|
150
|
+
puts format(" kernel, comparisons in registers %7.0f ms %.0fx",
|
|
151
|
+
network * 1e3, interpreted / network)
|
|
152
|
+
puts format(" the workspace is %d x %d doubles, which is %.0f MB the other one never touches",
|
|
153
|
+
SAMPLES, WINDOW, SAMPLES * WINDOW * 8 / 1048576.0)
|
|
@@ -0,0 +1,220 @@
|
|
|
1
|
+
# Lifting a parcel up a sounding: CAPE, CIN, and where the cloud starts.
|
|
2
|
+
#
|
|
3
|
+
# Take the air at the ground and lift it. It cools one way while it is dry
|
|
4
|
+
# and another way once it has condensed, and the pressure where it switches
|
|
5
|
+
# -- the lifting condensation level -- is not a constant: it is worked out
|
|
6
|
+
# from that column's temperature and dew point. Above it the parcel follows
|
|
7
|
+
# a moist adiabat, which has no closed form and is integrated step by step,
|
|
8
|
+
# each step starting from where the last one ended. Whether the parcel is
|
|
9
|
+
# warmer than the air around it decides whether that layer adds to the
|
|
10
|
+
# convective available potential energy or to the inhibition, and the level
|
|
11
|
+
# where the sign first turns positive -- the level of free convection -- is
|
|
12
|
+
# itself discovered on the way up.
|
|
13
|
+
#
|
|
14
|
+
# There is no expression over whole arrays for any of that. The state is
|
|
15
|
+
# carried up the column, the equation changes partway at a level each column
|
|
16
|
+
# picks for itself, and what a layer contributes depends on something that is
|
|
17
|
+
# not known until the walk reaches it. So a sounding is climbed in a loop,
|
|
18
|
+
# one column at a time, and that is exactly what a kernel is.
|
|
19
|
+
#
|
|
20
|
+
# The physics is the textbook version: Bolton's LCL, a pseudoadiabat
|
|
21
|
+
# integrated in pressure, and buoyancy from temperature rather than virtual
|
|
22
|
+
# temperature. Enough to be recognisable, not a substitute for a library.
|
|
23
|
+
#
|
|
24
|
+
# ruby examples/applications/parcel_ascent.rb
|
|
25
|
+
|
|
26
|
+
$LOAD_PATH.unshift(File.expand_path("../../lib", __dir__))
|
|
27
|
+
require "carray/jit"
|
|
28
|
+
|
|
29
|
+
RD = 287.05 # J/kg/K, dry air
|
|
30
|
+
CPD = 1005.7 # J/kg/K
|
|
31
|
+
LV = 2.501e6 # J/kg, vaporisation
|
|
32
|
+
EPS = 0.622 # molecular weight ratio
|
|
33
|
+
KAPPA = 0.2854 # Rd/cp
|
|
34
|
+
|
|
35
|
+
LEVELS = 91 # 1000 hPa to 100 hPa, every 10
|
|
36
|
+
COLUMNS = 4_000 # soundings to climb
|
|
37
|
+
SUBSTEPS = 4 # moist steps between two levels
|
|
38
|
+
|
|
39
|
+
pressure = CArray.double(LEVELS).seq!(1000.0, -10.0)
|
|
40
|
+
|
|
41
|
+
# Synthetic soundings: a lapse rate of 6.5 K/km over a surface that varies
|
|
42
|
+
# from column to column, isothermal above the tropopause.
|
|
43
|
+
random = CArray::Rng.new(seed: 20260913)
|
|
44
|
+
surface = CArray.double(COLUMNS)
|
|
45
|
+
surface.random!(rng: random)
|
|
46
|
+
surface = surface * 12.0 + 22.0 + 273.15 # 22 to 34 degC
|
|
47
|
+
spread = CArray.double(COLUMNS)
|
|
48
|
+
spread.random!(rng: random)
|
|
49
|
+
spread = spread * 14.0 + 2.0 # dew point depression
|
|
50
|
+
|
|
51
|
+
environment = CArray.double(COLUMNS, LEVELS)
|
|
52
|
+
height = CArray.double(LEVELS)
|
|
53
|
+
LEVELS.times { |k| height[k] = 44330.0 * (1.0 - (pressure[k] / 1013.25) ** 0.1903) }
|
|
54
|
+
CArray.jit_for(COLUMNS, LEVELS) { |c, k|
|
|
55
|
+
lapsed = surface[c] - 0.0065 * height[k]
|
|
56
|
+
environment[c, k] = lapsed < 216.65 ? 216.65 : lapsed
|
|
57
|
+
}
|
|
58
|
+
dewpoint = environment[nil, 0] - spread
|
|
59
|
+
|
|
60
|
+
cape = CArray.double(COLUMNS)
|
|
61
|
+
cin = CArray.double(COLUMNS)
|
|
62
|
+
lcl = CArray.double(COLUMNS)
|
|
63
|
+
lfc = CArray.double(COLUMNS)
|
|
64
|
+
el = CArray.double(COLUMNS)
|
|
65
|
+
|
|
66
|
+
def climb (pressure, environment, dewpoint, cape, cin, lcl, lfc, el)
|
|
67
|
+
columns, levels = environment.dim
|
|
68
|
+
CArray.jit_for(columns) { |c|
|
|
69
|
+
base = pressure[0]
|
|
70
|
+
start = environment[c, 0]
|
|
71
|
+
dew = dewpoint[c] < start ? dewpoint[c] : start
|
|
72
|
+
|
|
73
|
+
# Bolton (1980): the temperature the parcel reaches when it saturates,
|
|
74
|
+
# and the pressure that goes with it along the dry adiabat.
|
|
75
|
+
saturating = 1.0 / (1.0 / (dew - 56.0) + Math.log(start / dew) / 800.0) + 56.0
|
|
76
|
+
condensation = base * Math.exp(Math.log(saturating / start) / KAPPA)
|
|
77
|
+
lcl[c] = condensation
|
|
78
|
+
|
|
79
|
+
parcel = start
|
|
80
|
+
at = base
|
|
81
|
+
positive = 0.0
|
|
82
|
+
pending = 0.0 # negative area, until an LFC claims it
|
|
83
|
+
free = -1.0
|
|
84
|
+
equilibrium = -1.0
|
|
85
|
+
|
|
86
|
+
k = 1
|
|
87
|
+
while k < levels
|
|
88
|
+
target = pressure[k]
|
|
89
|
+
# Lift to this level, in at most two goes: dry as far as the
|
|
90
|
+
# condensation level, moist from there on.
|
|
91
|
+
while at > target + 1.0e-9
|
|
92
|
+
stop = at > condensation && condensation > target ? condensation : target
|
|
93
|
+
if at > condensation - 1.0e-9
|
|
94
|
+
parcel = start * Math.exp(KAPPA * Math.log(stop / base))
|
|
95
|
+
at = stop
|
|
96
|
+
else
|
|
97
|
+
step = (stop - at) / SUBSTEPS
|
|
98
|
+
n = 0
|
|
99
|
+
while n < SUBSTEPS
|
|
100
|
+
celsius = parcel - 273.15
|
|
101
|
+
saturation = 6.112 * Math.exp(17.67 * celsius / (celsius + 243.5))
|
|
102
|
+
mixing = EPS * saturation / (at - saturation)
|
|
103
|
+
numerator = RD * parcel + LV * mixing
|
|
104
|
+
denominator = CPD + (LV * LV * mixing * EPS) / (RD * parcel * parcel)
|
|
105
|
+
parcel = parcel + (numerator / denominator) * step / at
|
|
106
|
+
at = at + step
|
|
107
|
+
n = n + 1
|
|
108
|
+
end
|
|
109
|
+
end
|
|
110
|
+
end
|
|
111
|
+
|
|
112
|
+
# What this layer is worth.
|
|
113
|
+
excess = parcel - environment[c, k]
|
|
114
|
+
layer = RD * excess * Math.log(pressure[k - 1] / pressure[k])
|
|
115
|
+
if excess > 0.0
|
|
116
|
+
free = pressure[k] if free < 0.0
|
|
117
|
+
positive = positive + layer
|
|
118
|
+
equilibrium = pressure[k]
|
|
119
|
+
elsif free < 0.0
|
|
120
|
+
# Below the level of free convection, so this layer is what the
|
|
121
|
+
# parcel has to be pushed through: inhibition, not energy.
|
|
122
|
+
pending = pending + layer
|
|
123
|
+
end
|
|
124
|
+
k = k + 1
|
|
125
|
+
end
|
|
126
|
+
|
|
127
|
+
cape[c] = positive
|
|
128
|
+
cin[c] = free < 0.0 ? 0.0 : pending
|
|
129
|
+
lfc[c] = free
|
|
130
|
+
el[c] = equilibrium
|
|
131
|
+
}
|
|
132
|
+
end
|
|
133
|
+
|
|
134
|
+
climb(pressure, environment, dewpoint, cape, cin, lcl, lfc, el)
|
|
135
|
+
|
|
136
|
+
puts format("%d soundings, %d levels each", COLUMNS, LEVELS)
|
|
137
|
+
puts " T Td LCL LFC EL CAPE CIN"
|
|
138
|
+
[0, 1, 2, 3, 4].each do |c|
|
|
139
|
+
puts format(" %4.1f %4.1f %6.1f %6.1f %6.1f %7.1f %7.1f",
|
|
140
|
+
environment[c, 0] - 273.15, dewpoint[c] - 273.15,
|
|
141
|
+
lcl[c], lfc[c], el[c], cape[c], cin[c])
|
|
142
|
+
end
|
|
143
|
+
|
|
144
|
+
# What has to hold if the climb was a climb.
|
|
145
|
+
reached = lfc.gt(0.0)
|
|
146
|
+
puts
|
|
147
|
+
puts format(" %d of %d columns reached a level of free convection", reached.count(1), COLUMNS)
|
|
148
|
+
puts format(" the LFC is never below the LCL %s",
|
|
149
|
+
(lfc[reached] - lcl[reached]).max <= 0.0)
|
|
150
|
+
puts format(" the EL is never below the LFC %s",
|
|
151
|
+
(el[reached] - lfc[reached]).max <= 0.0)
|
|
152
|
+
puts format(" CAPE is zero exactly where there is no LFC %s",
|
|
153
|
+
cape[reached.not].max == 0.0)
|
|
154
|
+
puts format(" strongest column: CAPE %.0f J/kg with CIN %.0f J/kg",
|
|
155
|
+
cape.max, cin[cape.max_addr])
|
|
156
|
+
|
|
157
|
+
# The same climb in Ruby.
|
|
158
|
+
def climb_in_ruby (pressure, environment, dewpoint)
|
|
159
|
+
columns, levels = environment.dim
|
|
160
|
+
levels_a = pressure.to_a
|
|
161
|
+
cape = Array.new(columns, 0.0)
|
|
162
|
+
columns.times do |c|
|
|
163
|
+
base = levels_a[0]
|
|
164
|
+
start = environment[c, 0]
|
|
165
|
+
dew = [dewpoint[c], start].min
|
|
166
|
+
saturating = 1.0 / (1.0 / (dew - 56.0) + Math.log(start / dew) / 800.0) + 56.0
|
|
167
|
+
condensation = base * Math.exp(Math.log(saturating / start) / KAPPA)
|
|
168
|
+
parcel = start
|
|
169
|
+
at = base
|
|
170
|
+
positive = 0.0
|
|
171
|
+
free = -1.0
|
|
172
|
+
(1...levels).each do |k|
|
|
173
|
+
target = levels_a[k]
|
|
174
|
+
while at > target + 1.0e-9
|
|
175
|
+
stop = at > condensation && condensation > target ? condensation : target
|
|
176
|
+
if at > condensation - 1.0e-9
|
|
177
|
+
parcel = start * Math.exp(KAPPA * Math.log(stop / base))
|
|
178
|
+
at = stop
|
|
179
|
+
else
|
|
180
|
+
step = (stop - at) / SUBSTEPS
|
|
181
|
+
SUBSTEPS.times do
|
|
182
|
+
celsius = parcel - 273.15
|
|
183
|
+
saturation = 6.112 * Math.exp(17.67 * celsius / (celsius + 243.5))
|
|
184
|
+
mixing = EPS * saturation / (at - saturation)
|
|
185
|
+
numerator = RD * parcel + LV * mixing
|
|
186
|
+
denominator = CPD + (LV * LV * mixing * EPS) / (RD * parcel * parcel)
|
|
187
|
+
parcel += (numerator / denominator) * step / at
|
|
188
|
+
at += step
|
|
189
|
+
end
|
|
190
|
+
end
|
|
191
|
+
end
|
|
192
|
+
excess = parcel - environment[c, k]
|
|
193
|
+
if excess > 0.0
|
|
194
|
+
free = levels_a[k] if free < 0.0
|
|
195
|
+
positive += RD * excess * Math.log(levels_a[k - 1] / levels_a[k])
|
|
196
|
+
end
|
|
197
|
+
end
|
|
198
|
+
cape[c] = positive
|
|
199
|
+
end
|
|
200
|
+
cape
|
|
201
|
+
end
|
|
202
|
+
|
|
203
|
+
in_ruby = climb_in_ruby(pressure, environment, dewpoint)
|
|
204
|
+
agree = (0...COLUMNS).all? { |c| (cape[c] - in_ruby[c]).abs < 1.0e-9 }
|
|
205
|
+
puts format(" agrees with the same climb in Ruby %s", agree)
|
|
206
|
+
|
|
207
|
+
def timed (repeats = 3)
|
|
208
|
+
started = Process.clock_gettime(Process::CLOCK_MONOTONIC)
|
|
209
|
+
repeats.times { yield }
|
|
210
|
+
(Process.clock_gettime(Process::CLOCK_MONOTONIC) - started) / repeats
|
|
211
|
+
end
|
|
212
|
+
|
|
213
|
+
compiled = timed { climb(pressure, environment, dewpoint, cape, cin, lcl, lfc, el) }
|
|
214
|
+
interpreted = timed(1) { climb_in_ruby(pressure, environment, dewpoint) }
|
|
215
|
+
|
|
216
|
+
puts
|
|
217
|
+
puts format(" %6.1f ms compiled, %6.0f ms in Ruby (%.0fx)",
|
|
218
|
+
compiled * 1e3, interpreted * 1e3, interpreted / compiled)
|
|
219
|
+
puts format(" each column: %d levels, a condensation level of its own, and %d moist steps above it",
|
|
220
|
+
LEVELS, SUBSTEPS)
|
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
# Which of a million points are inside a polygon.
|
|
2
|
+
#
|
|
3
|
+
# The test is the old one: send a ray out from the point and count the edges
|
|
4
|
+
# it crosses; an odd count means inside. Counting is a loop over the edges,
|
|
5
|
+
# and the count is one integer that belongs to the point.
|
|
6
|
+
#
|
|
7
|
+
# An array expression cannot keep that integer. It can compare a point
|
|
8
|
+
# against an edge -- but the comparison has to be made for every point and
|
|
9
|
+
# every edge, and holding those answers is an array of points by edges. With
|
|
10
|
+
# a million points and two hundred edges that is a temporary of 1.5 GB, built
|
|
11
|
+
# so that it can immediately be summed away. The kernel keeps the count in a
|
|
12
|
+
# register and the polygon is read where it lies.
|
|
13
|
+
#
|
|
14
|
+
# ruby examples/applications/point_in_polygon.rb
|
|
15
|
+
|
|
16
|
+
$LOAD_PATH.unshift(File.expand_path("../../lib", __dir__))
|
|
17
|
+
require "carray/jit"
|
|
18
|
+
|
|
19
|
+
EDGES = 200
|
|
20
|
+
POINTS = 1_000_000
|
|
21
|
+
|
|
22
|
+
# A lobed blob, closed, going round once.
|
|
23
|
+
corner_x = CArray.double(EDGES)
|
|
24
|
+
corner_y = CArray.double(EDGES)
|
|
25
|
+
CArray.jit_for(EDGES) { |k|
|
|
26
|
+
angle = 2.0 * Math::PI * k / EDGES
|
|
27
|
+
radius = 1.0 + 0.35 * Math.sin(7.0 * angle) + 0.15 * Math.cos(3.0 * angle)
|
|
28
|
+
corner_x[k] = radius * Math.cos(angle)
|
|
29
|
+
corner_y[k] = radius * Math.sin(angle)
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
random = CArray::Rng.new(seed: 20260913)
|
|
33
|
+
point_x = CArray.double(POINTS)
|
|
34
|
+
point_x.random!(rng: random)
|
|
35
|
+
point_x = point_x * 3.2 - 1.6
|
|
36
|
+
point_y = CArray.double(POINTS)
|
|
37
|
+
point_y.random!(rng: random)
|
|
38
|
+
point_y = point_y * 3.2 - 1.6
|
|
39
|
+
|
|
40
|
+
inside = CArray.int8(POINTS)
|
|
41
|
+
|
|
42
|
+
def classify (point_x, point_y, corner_x, corner_y, inside)
|
|
43
|
+
edges = corner_x.elements
|
|
44
|
+
CArray.jit_for(point_x.elements) { |i|
|
|
45
|
+
x = point_x[i]
|
|
46
|
+
y = point_y[i]
|
|
47
|
+
crossings = 0
|
|
48
|
+
k = 0
|
|
49
|
+
while k < edges
|
|
50
|
+
j = k == 0 ? edges - 1 : k - 1
|
|
51
|
+
this_y = corner_y[k]
|
|
52
|
+
last_y = corner_y[j]
|
|
53
|
+
# The edge has to straddle the ray for the crossing to count, and the
|
|
54
|
+
# test is written so that a vertex exactly on the ray counts once.
|
|
55
|
+
if (this_y > y) != (last_y > y)
|
|
56
|
+
cut = corner_x[j] + (y - last_y) * (corner_x[k] - corner_x[j]) / (this_y - last_y)
|
|
57
|
+
crossings = crossings + 1 if x < cut
|
|
58
|
+
end
|
|
59
|
+
k = k + 1
|
|
60
|
+
end
|
|
61
|
+
inside[i] = crossings % 2
|
|
62
|
+
}
|
|
63
|
+
end
|
|
64
|
+
|
|
65
|
+
classify(point_x, point_y, corner_x, corner_y, inside)
|
|
66
|
+
|
|
67
|
+
puts format("%d points against a polygon of %d edges", POINTS, EDGES)
|
|
68
|
+
puts format(" inside %d (%.1f%% of the square they were drawn in)",
|
|
69
|
+
inside.sum, 100.0 * inside.sum / POINTS)
|
|
70
|
+
|
|
71
|
+
# The same test in Ruby, over as many points as is polite to wait for.
|
|
72
|
+
SAMPLE = 20_000
|
|
73
|
+
|
|
74
|
+
def classify_in_ruby (x, y, corner_x, corner_y)
|
|
75
|
+
edges = corner_x.elements
|
|
76
|
+
odd = false
|
|
77
|
+
j = edges - 1
|
|
78
|
+
(0...edges).each do |k|
|
|
79
|
+
if (corner_y[k] > y) != (corner_y[j] > y)
|
|
80
|
+
cut = corner_x[j] + (y - corner_y[j]) *
|
|
81
|
+
(corner_x[k] - corner_x[j]) / (corner_y[k] - corner_y[j])
|
|
82
|
+
odd = !odd if x < cut
|
|
83
|
+
end
|
|
84
|
+
j = k
|
|
85
|
+
end
|
|
86
|
+
odd ? 1 : 0
|
|
87
|
+
end
|
|
88
|
+
|
|
89
|
+
def timed (repeats = 3)
|
|
90
|
+
started = Process.clock_gettime(Process::CLOCK_MONOTONIC)
|
|
91
|
+
repeats.times { yield }
|
|
92
|
+
(Process.clock_gettime(Process::CLOCK_MONOTONIC) - started) / repeats
|
|
93
|
+
end
|
|
94
|
+
|
|
95
|
+
agree = (0...SAMPLE).all? { |i|
|
|
96
|
+
inside[i] == classify_in_ruby(point_x[i], point_y[i], corner_x, corner_y)
|
|
97
|
+
}
|
|
98
|
+
puts format(" agrees with Ruby over the first %d %s", SAMPLE, agree)
|
|
99
|
+
|
|
100
|
+
compiled = timed { classify(point_x, point_y, corner_x, corner_y, inside) }
|
|
101
|
+
interpreted = timed(1) {
|
|
102
|
+
SAMPLE.times { |i| classify_in_ruby(point_x[i], point_y[i], corner_x, corner_y) }
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
puts
|
|
106
|
+
puts format(" %6.0f ms here for %d points", compiled * 1e3, POINTS)
|
|
107
|
+
puts format(" %6.0f ms in Ruby for %d, so about %.0f s for the million",
|
|
108
|
+
interpreted * 1e3, SAMPLE, interpreted * POINTS / SAMPLE)
|
|
109
|
+
puts format(" the array route would hold %d x %d doubles on the way -- %.1f GB,",
|
|
110
|
+
POINTS, EDGES, POINTS.to_f * EDGES * 8 / (1 << 30))
|
|
111
|
+
puts " built only to be summed away; this one holds one integer, in a register"
|