farm.rb 0.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +7 -0
- data/CHANGELOG +22 -0
- data/Gemfile +3 -0
- data/LICENSE +21 -0
- data/MEASUREMENTS.md +246 -0
- data/README.md +80 -0
- data/ROADMAP.md +64 -0
- data/Rakefile +9 -0
- data/farm.rb.gemspec +48 -0
- data/lib/Farm/NotParallelisable.rb +7 -0
- data/lib/Farm/Probe.rb +60 -0
- data/lib/Farm/RactorExecutor.rb +75 -0
- data/lib/Farm/SerialExecutor.rb +28 -0
- data/lib/Farm/VERSION.rb +6 -0
- data/lib/farm.rb +43 -0
- data/test/Farm/Probe_test.rb +56 -0
- data/test/Farm/RactorExecutor_test.rb +62 -0
- data/test/Farm/SerialExecutor_test.rb +43 -0
- data/test/Farm/VERSION_test.rb +21 -0
- data/test/farm_test.rb +67 -0
- data/test/gemspec_test.rb +33 -0
- data/test/loading_test.rb +28 -0
- metadata +103 -0
checksums.yaml
ADDED
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
---
|
|
2
|
+
SHA256:
|
|
3
|
+
metadata.gz: 42bdfdd82e47daa951ea8607257d95bb223f070cc2decbb9aafe931c7921b8f2
|
|
4
|
+
data.tar.gz: 46a18d3d16da99e59273b2c15c2cf9109e8cc4c009a1fbecb32916e859c2e55c
|
|
5
|
+
SHA512:
|
|
6
|
+
metadata.gz: adaef059970b22e34b553e5d981433b1b13d4290e1270501c4cff5f6c451d60614afbf10a288ff345cc42b489fc97b426751a19d42c56bd8d591d98a40e38e06
|
|
7
|
+
data.tar.gz: e8a26416d0d8002838fcb4998f006adc26eaee2cefe0fc695d1e73371fcb5c8e0907bc146a0c49b106ac9e84696ef2ab0525d33c6d15a2f29d045945c716eaf9
|
data/CHANGELOG
ADDED
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
# CHANGELOG
|
|
2
|
+
|
|
3
|
+
## 20261004
|
|
4
|
+
|
|
5
|
+
0.0.0: + Farm, + Farm.each, map, Farm::Probe, Farm::RactorExecutor, Farm::SerialExecutor
|
|
6
|
+
|
|
7
|
+
1. + Farm.map and Farm.each: the whole API. Work travels as a name — a callable and a method upon it, #call being the default — so that it can cross to a worker where a block could not. Blocks are accepted and run serially, there being nothing to name.
|
|
8
|
+
2. + Farm::RactorExecutor: a pool of ractors, as wide as the machine has performance cores, fed round-robin with the results coming back over a single port and put back in order. Each worker rescues its own failures and tags them, a worker which died delivering nothing to the port and leaving its replacement waiting upon it.
|
|
9
|
+
3. + Farm::SerialExecutor: the ordinary way of doing work, and what every executor is measured against and falls back to.
|
|
10
|
+
4. + Farm::NotParallelisable: raised by an executor when the work could not cross, upon which Farm runs the same work serially rather than surfacing the failure. Speculate and fall back: try the machine, and take the slow correct answer over none.
|
|
11
|
+
5. + Farm::Probe: what the machine says about itself — logical and performance cores, total and resident memory, load averages — the measurements later versions will decide substrate and width from. Each figure is asked of the platform rather than tabulated, and platforms which do not answer get a nil rather than a guess.
|
|
12
|
+
6. spec.required_ruby_version = '>= 4.0': the executor is written against Ruby 4's Ractor, where #take is gone in favour of #value and Ractor::Port is the queue.
|
|
13
|
+
7. + ROADMAP.md: the 0.1.0 direction — the send-time rescue gap closed; fork and thread executors beside the ractor pool, falling back by class; the job-level measurement loop; an in-process wisdom — and the decisions which hold across versions: work travels as a name or not at all; Farm embeddable in both directions beneath a consumer; every executor observationally equivalent to serial but for time; width, substrate and threshold measured, never tabulated.
|
|
14
|
+
8. + MEASUREMENTS.md: the measured basis for those decisions and the constants the measurement loop starts from — substrate costs and scaling sweeps, boundary-carrying costs, the isolation and reachability surveys, which callable shapes cross and why, fork's memory-bound cost, the threshold algebra, the one hybrid experiment and what its inconclusiveness implies, prior art worth stealing, and the five instruments which lied. Every figure is a prior upon one machine's two days, with its spread, to be corrected in situ.
|
|
15
|
+
|
|
16
|
+
## 20260822
|
|
17
|
+
|
|
18
|
+
+ measurements/: the parallelism note (20260809–22; ruby 4.0.5; Apple silicon) and the fourteen scripts whose runs produced its figures, imported verbatim. The scripts are the instruments and the note is their record.
|
|
19
|
+
|
|
20
|
+
+ MEASUREMENTS.md: the note abridged for the repo's own use — what was measured, what it decided, and the constants the measurement loop starts from, with every figure a prior from one machine to be corrected in situ.
|
|
21
|
+
|
|
22
|
+
+ measurements/README.md: what the archive is and how to run it — provenance, any script runnable alone, and the warning that the spread is the finding. Nothing in measurements/ ships in the gem.
|
data/Gemfile
ADDED
data/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 thoran
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
data/MEASUREMENTS.md
ADDED
|
@@ -0,0 +1,246 @@
|
|
|
1
|
+
# Farm Measurements
|
|
2
|
+
|
|
3
|
+
Date: 20261004
|
|
4
|
+
|
|
5
|
+
The measured basis for the decisions in ROADMAP.md and for the constants the measurement loop starts from. Everything here was measured on one machine: an Apple silicon Mac, Darwin 24.6.0, twelve logical cores — eight performance and four efficiency — 96GB of memory, ruby 4.0.5, over two days in August 2026. Run-to-run spread is around forty-five per cent for `fork` and much tighter for the cheaper substrates. No figure here is a constant: these are priors with noise floors, to be corrected upon the machine in hand. What may be compared with what is stated beside each table, since several of these tables first told a wrong story by quoting against the wrong baseline.
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
## Substrate startup cost
|
|
9
|
+
|
|
10
|
+
The round trip for one unit doing nothing — start the worker, hand it a payload where stated, receive the answer where offered, wait for exit. What a job pays purely for being parallel, before any work:
|
|
11
|
+
|
|
12
|
+
| substrate | per unit |
|
|
13
|
+
| --- | --- |
|
|
14
|
+
| Thread, no result | 0.042 ms |
|
|
15
|
+
| Thread, 64B result | 0.032 ms |
|
|
16
|
+
| Ractor, 64B result | 0.033 ms |
|
|
17
|
+
| fork, no result | 1.004 ms |
|
|
18
|
+
| fork + pipe, 64B result | 1.129 ms |
|
|
19
|
+
| fork + pipe, 1MB result | 3.974 ms |
|
|
20
|
+
| spawn `/usr/bin/true` | 2.602 ms |
|
|
21
|
+
|
|
22
|
+
- A Ractor costs what a thread costs to start, and thirty times less than a fork. The objection to Ractors is the isolation rules, not the price of admission.
|
|
23
|
+
- `fork` is a millisecond: cheap in absolute terms, dear in relative ones — fine for units of tens of milliseconds and up, ruinous for units of one.
|
|
24
|
+
- The result dominates once large: 64B adds 0.125ms to a fork; 1MB adds 2.97ms. The size of what comes back is an input to the decision, not an afterthought.
|
|
25
|
+
- `spawn` is two to three forks *for a trivial program*, and the multiple is not its own — see fork vs spawn below.
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
## Scaling
|
|
29
|
+
|
|
30
|
+
### CPU-bound
|
|
31
|
+
|
|
32
|
+
Twenty-four units of one arithmetic unit of work — touching nothing but an Integer constant, so the Ractor column measures scaling rather than isolation rules — batched at each width. Speedups are against each substrate's own single-worker row, which still paid that substrate's per-unit setup, so width alone is what the figure shows. Wall times were comparable across the three and Ractors were quickest at every width.
|
|
33
|
+
|
|
34
|
+
| width | Ractor | Thread | fork |
|
|
35
|
+
| --- | --- | --- | --- |
|
|
36
|
+
| 1 | 1.00x | 1.00x | 1.00x |
|
|
37
|
+
| 2 | 1.75x | 0.97x | 1.74x |
|
|
38
|
+
| 4 | 3.33x | 0.98x | 2.42x |
|
|
39
|
+
| 6 | 4.13x | 0.91x | 3.13x |
|
|
40
|
+
| 8 | 4.82x | 0.98x | 3.96x |
|
|
41
|
+
| 12 | 4.49x | 0.96x | 4.70x |
|
|
42
|
+
| 16 | 5.08x | 0.72x | 5.17x |
|
|
43
|
+
|
|
44
|
+
- Threads gain nothing upon computation — never above the single-worker time, 0.72x where contention starts. Not slightly less: nothing. The GVL is held throughout.
|
|
45
|
+
- Ractors were quicker than `fork` in absolute time at every width, most in the middle (0.092s against 0.149s at four, 0.064s against 0.091s at eight).
|
|
46
|
+
- Scaling flattens somewhere past eight and the exact width stops mattering. Take the performance-core count because the returns have gone — and because the job threshold rises with width — not because wider hurts. The dip at twelve (4.82 → 4.49 → 5.08) is an eight-per-cent spread upon a machine which cannot be trusted to report five: read it as flat. A "regression at twelve" reported earlier in this work was an artefact of measuring `system` — fork plus exec of Ruby, forty milliseconds a unit — rather than fork.
|
|
47
|
+
|
|
48
|
+
### Waiting
|
|
49
|
+
|
|
50
|
+
Forty-eight units of a 0.02s sleep, swept wider since a waiting worker occupies no core and the optimum is not bounded by core count.
|
|
51
|
+
|
|
52
|
+
| width | Ractor | Thread | fork |
|
|
53
|
+
| --- | --- | --- | --- |
|
|
54
|
+
| 1 | 1.00x | 1.00x | 1.00x |
|
|
55
|
+
| 2 | 2.01x | 1.94x | 2.09x |
|
|
56
|
+
| 4 | 3.95x | 3.82x | 4.06x |
|
|
57
|
+
| 8 | 7.69x | 7.44x | 7.67x |
|
|
58
|
+
| 16 | 15.64x | 15.10x | 12.72x |
|
|
59
|
+
| 24 | 22.97x | 21.69x | 13.51x |
|
|
60
|
+
| 48 | 46.85x | 42.68x | 19.97x |
|
|
61
|
+
|
|
62
|
+
- Ractors win here too — 98% of perfect scaling at forty-eight against threads' 89% — and quicker in absolute time at every width. This contradicted advice written before it was measured; threads are the *best* answer for no workload class measured. They are the fall-back for waiting work exactly as fork is the fall-back for computation: what you use when isolation refuses.
|
|
63
|
+
- `fork` scales with the field to eight and then stalls — 19.97x at forty-eight where the others are near-perfect. Forty-eight processes is a great deal of machinery to have sitting idle.
|
|
64
|
+
- Unswept past forty-eight Ractors. Genuinely IO-bound work often wants hundreds or thousands of concurrent waits, and a Ractor carries a native thread apiece; that territory belongs to fibers, which are unmeasured here.
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
## Crossing a Ractor boundary
|
|
68
|
+
|
|
69
|
+
A payload in and the same payload out for one unit, three ways over, with fork-and-pipe for scale:
|
|
70
|
+
|
|
71
|
+
| payload | ractor copy | ractor move | ractor shared | fork + pipe |
|
|
72
|
+
| --- | --- | --- | --- | --- |
|
|
73
|
+
| 1KB | 0.033 ms | 0.019 ms | 0.016 ms | 0.810 ms |
|
|
74
|
+
| 100KB | 0.028 ms | 0.030 ms | 0.023 ms | 1.271 ms |
|
|
75
|
+
| 1MB | 0.145 ms | 0.184 ms | 0.036 ms | 3.846 ms |
|
|
76
|
+
| 10MB | 0.391 ms | 0.426 ms | 0.028 ms | 26.304 ms |
|
|
77
|
+
|
|
78
|
+
1. Ractors beat fork-and-pipe on transfer at every size, by twenty-four to sixty-seven times down these rows. Read the floor — about 24x — rather than the trend; the middle rows carry their own noise at this scale.
|
|
79
|
+
2. Shared does not follow the payload: 0.016–0.036ms while the payload grows ten-thousandfold. Nothing is copied, so size is not what it is paying for. It is the only arrangement here where payload size drops out of the decision.
|
|
80
|
+
3. Moving is not the win it sounds like: forty-two per cent cheaper than copying at a kilobyte, then worse at every size above. Reach for move to enforce ownership, not for speed.
|
|
81
|
+
4. `fork` moves about 380MB/s through `Marshal` and a pipe — the rate rises with size as the ~0.81ms fixed cost amortises; a megabyte moved 329–337MB/s, ten megabytes 392MB/s. A megabyte of result is roughly three milliseconds of pure overhead by pipe against 0.145ms by Ractor. Result size can choose the substrate on its own.
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
## What crosses
|
|
85
|
+
|
|
86
|
+
Twenty-six objects surveyed — plain data, stdlib values, library instances, things holding resources:
|
|
87
|
+
|
|
88
|
+
**21 shareable, 21 copyable across a Ractor, 18 marshallable, 1 froze a global.**
|
|
89
|
+
|
|
90
|
+
- Every real payload passed — parsed JSON and YAML, Time, Date, Pathname, URI, Set, Range, Regexp, arrays of hashes, library instances. Isolation is not the obstacle it is reputed to be *for data*, which is what a dispatched unit usually carries.
|
|
91
|
+
- Ractors are more permissive than `Marshal` (21 against 18): a Struct instance, a File handle and an object holding `STDOUT` all cross a Ractor boundary and none survives Marshal. On this sample there is nothing a process can be sent which a Ractor cannot.
|
|
92
|
+
- The failures are all code or resources, never data: Proc, Method, Thread, StringIO, Logger, Exception. (A Logger holds a Monitor, and a mutex is not shareable — the common case of an object guarded by a lock fails safely. An earlier construction, `Logger.new(File::NULL)`, had flattered the survey by holding no real handle.)
|
|
93
|
+
- The three outcomes, which matter more than the tally:
|
|
94
|
+
1. **Refused, loudly** — Proc gives `Ractor::IsolationError`; Thread, StringIO, Exception and lock-carriers give `Ractor::Error`. Nothing damaged; the caller learns.
|
|
95
|
+
2. **Succeeds, harmlessly** — all the data above.
|
|
96
|
+
3. **Succeeds, and destroys what it touched** — a File handle comes out frozen and the next write raises; an object holding `STDOUT` froze `STDOUT` process-wide, and every subsequent `puts` raised `FrozenError`. No exception at the time; the damage surfaces later and elsewhere.
|
|
97
|
+
|
|
98
|
+
The hazard has a shape: **objects holding a raw IO succeed and break; objects holding a lock fail safely.** The destructive case is silent, no rescue can see it, and it is why the rule is never to call `make_shareable` upon data the library does not own — and why copying, at 0.03ms a kilobyte, stays the default.
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
## What blocks the run
|
|
102
|
+
|
|
103
|
+
Eight payloads, each reaching a resource indirectly — class variable, class-level instance variable, mutable constant, frozen constant, global variable, lazily built handle, closure over `STDOUT`, nested object with IO:
|
|
104
|
+
|
|
105
|
+
- **Five of eight passed every check at the boundary and then failed to run.** Isolation governs what the code *touches*, not only what the payload *holds*: class variables, class-level ivars, unshareable constants and globals are refused from a non-main Ractor whatever the payload looks like. `Ractor.shareable?(payload)` predicts almost nothing about whether the job will run — the obstruction is in the method body, not the argument.
|
|
106
|
+
- The failures are loud, which is the saving grace: every one raised `Ractor::IsolationError` inside, surfacing as `Ractor::RemoteError` outside. Suitability is discoverable by trying, hence speculate-and-fall-back. (The cost of a refused speculation was never timed and is assumed cheap — see Unanswered.)
|
|
107
|
+
- Two rows ran but should comfort nobody: a lazily built handle, because it is opened inside the Ractor — each gets its own, which is rather elegant; and a nested object with IO, only because the work asked the handle its class — `make_shareable` had silently frozen it two levels down, and the first write would have raised.
|
|
108
|
+
- `freeze` is not shareability: `['a', 'b'].freeze` freezes the array and leaves the strings mutable, and the constant is still refused. `Ractor.make_shareable` at the declaration runs. The constants problem is solvable at definition and repays itself almost immediately:
|
|
109
|
+
|
|
110
|
+
| size | make_shareable, once | read in main | read in Ractor | copied in, per dispatch |
|
|
111
|
+
| --- | --- | --- | --- | --- |
|
|
112
|
+
| 1k | 0.018 ms | 0.002 ms | 0.071 ms | 0.062 ms |
|
|
113
|
+
| 100k | 1.847 ms | 0.152 ms | 0.438 ms | 2.117 ms |
|
|
114
|
+
| 1M | 17.454 ms | 1.235 ms | 1.480 ms | 22.600 ms |
|
|
115
|
+
|
|
116
|
+
Reading a shareable million-element constant from a Ractor costs 1.480ms against handing a fresh copy in per dispatch at 22.600ms — the one-off 17.454ms is recovered in **one dispatch**. Class variables, class-level ivars and globals have no such fix; code which memoises into class-level state cannot run in a Ractor and falls back.
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
## fork and spawn are different things, and neither cost is a constant
|
|
120
|
+
|
|
121
|
+
`fork` duplicates this process; `spawn` forks and execs, throwing the duplicate away. Non-cost differences, which decide more often than the prices do: the forked child holds every object and all loaded code, but keeps only the forking thread — a mutex another thread held stays locked forever in the child; exec wipes that clean. And there is no fork on Windows.
|
|
122
|
+
|
|
123
|
+
**Exec scales with what is being loaded:**
|
|
124
|
+
|
|
125
|
+
| child | per spawn | the exec share |
|
|
126
|
+
| --- | --- | --- |
|
|
127
|
+
| bare fork, for the base | 0.988 ms | - |
|
|
128
|
+
| `/bin/echo` | 2.004 ms | 1.016 ms |
|
|
129
|
+
| `/usr/bin/true` | 2.602 ms | 1.614 ms |
|
|
130
|
+
| `ruby -e ''` | 40.393 ms | 39.405 ms |
|
|
131
|
+
|
|
132
|
+
Exec'ing a small C program costs a millisecond or two; exec'ing Ruby costs forty. `spawn` is for running some other program and is absurd for Ruby work per unit.
|
|
133
|
+
|
|
134
|
+
**Fork scales with the parent's memory, and with its high-water mark rather than its live set:**
|
|
135
|
+
|
|
136
|
+
| ballast | RSS | first fork | median of the rest |
|
|
137
|
+
| --- | --- | --- | --- |
|
|
138
|
+
| none | 14.4 MB | 1.436 ms | 1.158 ms |
|
|
139
|
+
| 64MB | 79.8 MB | 4.818 ms | 3.648 ms |
|
|
140
|
+
| 128MB | 144.4 MB | 4.720 ms | 3.212 ms |
|
|
141
|
+
| 256MB | 275.4 MB | 8.498 ms | 6.850 ms |
|
|
142
|
+
| 512MB | 535.1 MB | 12.236 ms | 8.233 ms |
|
|
143
|
+
| 1024MB | 1055.8 MB | 24.495 ms | 21.936 ms |
|
|
144
|
+
|
|
145
|
+
- A gigabyte held makes every fork twenty times dearer. Freeing it and collecting does not bring the cost back — 20.4ms held, 13.2ms after free and `GC.start` — because Ruby returns little to the operating system. A process which was once large stays expensive to fork from, permanently.
|
|
146
|
+
- The rate is roughly 2.0ms per 100MB of RSS, taken against the none-row baseline (subtracting the ~1.16ms fixed cost of any fork first; the rows read 3.81, 1.58, 2.18, 1.36 and 2.00, a spread of nearly three to one). Proportional it is not: taking 2.0 as *the* rate predicts other rows within about forty per cent. Enough to decide whether forking is plausibly worth trying; nowhere near enough to set the threshold that decides it.
|
|
147
|
+
- The first fork is dearer by twelve to forty-nine per cent — the direction is sound (all six rows agreed, compared within single runs), the size is not believed. A one-sample probe over-reports by up to half; it must discard one first, which doubles its cost.
|
|
148
|
+
- Probing at startup cost 8.5ms in a small process and **129.6ms in a process holding a gigabyte** — most expensive exactly where most wanted. So the earlier "measure fork at startup" instruction was wrong. **Instrument the first real fork instead**: RSS times the calibrated rate as the prior, correct from the first real dispatch, and carry the correction for the life of the process. Estimate-then-measure — the same shape as FFTW's planner, arrived at from the other direction.
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
## The thresholds
|
|
152
|
+
|
|
153
|
+
`N` units of work, each `D` in duration, setup `S` per worker, `W` workers, `k` the speedup actually achieved.
|
|
154
|
+
|
|
155
|
+
Per unit (a worker started per unit): worth it when `D > S / (k − 1)`.
|
|
156
|
+
Per job (a pool of `W` workers fed all `N` units — the question a pool actually faces, setup being paid `W` times rather than `N`): worth it when
|
|
157
|
+
|
|
158
|
+
N·D > W·S / (1 − 1/k)
|
|
159
|
+
|
|
160
|
+
Thresholds on total serial work, from the tables above:
|
|
161
|
+
|
|
162
|
+
| substrate | class | S | W | k | threshold |
|
|
163
|
+
| --- | --- | --- | --- | --- | --- |
|
|
164
|
+
| Ractor | computation | 0.033 ms | 8 | 4.82 | ~0.33 ms |
|
|
165
|
+
| fork | computation | 1.129 ms | 8 | 3.96 | ~12 ms |
|
|
166
|
+
| spawn | computation | 2.602 ms | 8 | 3.96 | ~28 ms |
|
|
167
|
+
| Ractor | waiting | 0.033 ms | 48 | 46.85 | ~1.6 ms |
|
|
168
|
+
| Thread | waiting | 0.032 ms | 48 | 42.68 | ~1.6 ms |
|
|
169
|
+
| fork | waiting | 1.129 ms | 48 | 19.97 | ~57 ms |
|
|
170
|
+
|
|
171
|
+
- Below the threshold for every substrate: run serially and stop.
|
|
172
|
+
- **Going wider raises the threshold** — 0.33ms at eight Ractors, 1.6ms at forty-eight, the setup being paid forty-eight times. A small job which would repay eight workers may not repay forty-eight; for fork, 12ms at eight against 57ms at forty-eight, which rules fork out for wide waiting work.
|
|
173
|
+
- The fork rows are for a small process: inside an application holding a gigabyte they are ~20ms and ~200ms.
|
|
174
|
+
- Three inputs, three different questions. Total job size decides *whether*; per-unit duration decides *how to slice* (units below the threshold are batched until the batch clears it); result size decides *which substrate*.
|
|
175
|
+
- Classifying is cheap and empirical: run the work at two widths and watch whether wall time falls. No scaling upon threads is CPU-bound (the GVL is held); scaling is waiting. That is a few milliseconds of probing, and no understanding of the work at all.
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
## Hybrid arrangements
|
|
179
|
+
|
|
180
|
+
The object of the library is combinations, and exactly one experiment exists: the same twenty-four units and the same total width arranged as P processes of R Ractors for every factoring of eight and twelve, three runs apiece.
|
|
181
|
+
|
|
182
|
+
- The first run said the hybrid wins by thirty-six per cent, with a clean mechanism ready to explain it. The next two runs: **the winner changes every run, and every arrangement varies by more than half its own value between runs** (spread 6–78%, within a single script, on one machine). The differences between arrangements are smaller than the differences between runs of the same arrangement. There is no result, and the thirty-six per cent was nothing.
|
|
183
|
+
- The experiment was also aimed at the wrong workload — a tight arithmetic loop allocating nothing is the case least likely to contend upon anything process-global, which was the conjecture being tested.
|
|
184
|
+
- What it implies for the search the library must eventually run: **measure the noise floor first** (it is per arrangement, not a constant — 6% and 78% sat in the same table); **treat "cannot distinguish" as a result** and take the simplest arrangement, usually the fewest processes; **prune before searching** — validity (one Ractor probe rules out every Ractor-bearing arrangement at once), threshold (rises with width, so small jobs rule out wide arrangements arithmetically), class (a thread-bearing arrangement is pointless for computation, settled by the two-width probe); **repeat until the leaders separate past the noise, or the budget runs out**; and **cache the answer against the shape of the work**. The difficulty is not the search, which is arithmetic. It is the statistics.
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
## Reaching past the machine
|
|
188
|
+
|
|
189
|
+
Nothing remote was timed; this section is reasoning, and named as such. The framing that survives contact is not local-versus-remote but whether the tuner's cost model stays true.
|
|
190
|
+
|
|
191
|
+
A substrate is **bounded** when four things hold: latency is small with a short tail, so a handful of samples describe it; failure is all-or-nothing and detected quickly; capacity cannot be withdrawn while work is in flight; and using it costs no money. Threads, Ractors and `fork` satisfy all four, as does a local daemon over a socket and a peer in the same chassis. Rented capacity satisfies none of them. The rule:
|
|
192
|
+
|
|
193
|
+
> The tuner may select freely among bounded substrates, measurement being a valid way to choose there. Unbounded substrates want an explicit opt-in, because measurement stops being predictive — and because of the standing principle that **an autotuner may spend time without asking, but it may not spend money without being told**. Time it can measure and recover from; money it cannot.
|
|
194
|
+
|
|
195
|
+
Three tiers of remote fall out:
|
|
196
|
+
|
|
197
|
+
- **A peer in the same chassis** — sub-millisecond, on a bus you own, at no price. Bounded on all four counts, and at under a millisecond *cheaper to reach than `fork` is*: the demonstration that the gradient is latency, not locality.
|
|
198
|
+
- **A peer on the LAN, discovered automatically** — low milliseconds, on hardware you own and supervise. Admissible with a wider tolerance and a liveness check, on the understanding that one bad switch turns it unbounded without warning.
|
|
199
|
+
- **Rented capacity** — revocable and billed. Behind the explicit opt-in.
|
|
200
|
+
|
|
201
|
+
Whatever transports carry the first two will want a settled vocabulary for how a binding ends — graceful close, expiry, withdrawal by the far side, silence past a heartbeat, non-payment — those being the states the policy acts upon, invariant across protocols.
|
|
202
|
+
|
|
203
|
+
Two things a remote substrate offers which no local one does:
|
|
204
|
+
|
|
205
|
+
- **Advertised capability**: a peer can say what it has, where a local substrate must be measured to be known. Take the advertisement as the prior, verify it under load, cache the correction against that peer — FFTW's `ESTIMATE` against `MEASURE`, and the right treatment of a configuration file or a user's hint besides.
|
|
206
|
+
- **Price**: everything local optimises time alone. A billed substrate makes the objective ambiguous — minimise time within a budget, or cost within a deadline — which is API-shaping, and worth deciding early.
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
## Prior art worth stealing
|
|
210
|
+
|
|
211
|
+
Not in Ruby, so far as is known. Elsewhere, repeatedly:
|
|
212
|
+
|
|
213
|
+
- **FFTW's planner** — nearest in spirit. Benchmarks several strategies at runtime, picks the fastest, caches the decision as "wisdom" persisting between runs; exposes planner effort as a dial — `ESTIMATE` guesses from a cost model, `MEASURE` tries alternatives, `EXHAUSTIVE` tries many. Steal the wisdom cache and the effort dial.
|
|
214
|
+
- **.NET's thread pool** — a hill-climbing controller since about 2009: varies the thread count, watches completion throughput, keeps changes which help. Online tuning rather than up-front calibration; the answer to workloads whose character changes partway. Steal *keep watching after the first decision*.
|
|
215
|
+
- **Intel TBB** — decides grain size via `auto_partitioner`, backed by work-stealing. Steal as the default scheduling strategy where units vary in cost; dealing units out evenly is only right when the units are alike.
|
|
216
|
+
- **OpenMP** — `schedule(auto)` beside `static`, `dynamic` and `guided`, plus `OMP_DYNAMIC`. Steal the API shape: automatic is the default, every knob remains sayable, and the manual strategies share the vocabulary.
|
|
217
|
+
- **Cilk** — the provably-efficient work-stealing scheduler the others build on.
|
|
218
|
+
- **ATLAS** and **OpenTuner** — the general autotuning tradition.
|
|
219
|
+
- **concurrent-ruby** — mechanism, not policy: thread-centric, pool sizing but nothing which adapts to measured behaviour, and nothing spanning processes or Ractors, which is where Ruby's real parallelism lives. Neither adopt wholesale nor refuse: policy and measurement stay Farm's own; mechanism is pluggable, with concurrent-ruby as one backend among Ractor and fork.
|
|
220
|
+
|
|
221
|
+
What is genuinely unclaimed is the Ruby-specific dimension. In Java, .NET and C++ the autotuners choose *width* and *granularity*, a thread being a thread. In Ruby the first decision is which substrate to run upon at all, and it changes the answer by an order of magnitude more than width does: the same computation gains nothing upon threads and 4.82x upon Ractors at the same width; the same waiting gains 46.85x upon Ractors against 19.97x upon processes. Substrate is worth more than width by an order of magnitude, and those languages have no GVL to work around.
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
## The instruments which lied
|
|
225
|
+
|
|
226
|
+
Five measurements in this body of work returned a plausible number for something other than what was being asked. Each was caught by the answer looking wrong rather than by the instrument complaining:
|
|
227
|
+
|
|
228
|
+
1. **Timing failures.** 0.02s per compile — impossible for a translation unit including `<regex>`; every compile was erroring instantly on a missing sysroot, and the harness was timing failures. The guard which survived: the calibration harness raises unless each unit returns truthy. Anything which times a stranger's work must *measure that the work happened*.
|
|
229
|
+
2. **Measuring `system` and calling it `fork`.** Forty milliseconds a unit of interpreter startup, inventing a regression at twelve workers which does not exist.
|
|
230
|
+
3. **Comparing the fix against itself.** The constants survey passed the already-shareable object as the "copied in" case; shareable objects cross by reference, so nothing was copied, and the alternative appeared cheaper than the fix.
|
|
231
|
+
4. **A Method built on the far side.** The callable survey constructed the Method inside the Ractor and reported every row as running — a local construction agreeing with itself, having crossed nothing.
|
|
232
|
+
5. **A checker quoting prose.** The staleness audit looked for a whole sentence as its marker; rewording the sentence made a properly-marked claim report as unmarked.
|
|
233
|
+
|
|
234
|
+
What they share is that **nothing failed**. A wrong instrument that raises is a nuisance; one that returns a number is a trap, because the number gets written down and reasoned from. The defence which worked every time was arithmetic done by hand upon the answer: 0.02s is too fast for `<regex>`; 22.6ms against 1.5ms is the wrong way round for a copy; a fork cannot cost 45% more than itself. Not verification of the code — verification of the answer.
|
|
235
|
+
|
|
236
|
+
Staleness itself has two shapes. A **figure** leaves arithmetic descendants — a constant feeding two formulas and five sentences wants all eight sites changed together, and correcting the table is not correcting the document. A **claim** leaves no arithmetic, only wording, and often the measurement behind it was sound and only the conclusion wrong — mark it and point at the correction rather than deleting it.
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
## Unanswered, and named rather than buried
|
|
240
|
+
|
|
241
|
+
- **Ractors past forty-eight workers**: a native thread apiece; hundreds-of-waits territory is fibers', which are unmeasured entirely.
|
|
242
|
+
- **How much real code satisfies the isolation rules**, as against how much real data: twenty-six payloads and eight routes are not a corpus, and a real application's dispatched work reaches configuration, connection pools and loggers by paths nobody enumerates.
|
|
243
|
+
- **The cost of a refused speculation**: speculate-and-fall-back assumes the failed attempt is cheap. Never timed; an `IsolationError` raised deep in a call stack may not be.
|
|
244
|
+
- **Anything remote**: none of it timed. The bounded/unbounded framing is reasoning.
|
|
245
|
+
- **Whether a hybrid beats either pure form**: measured and inconclusive; wants many more rounds, a quieter machine, and a workload which allocates.
|
|
246
|
+
- **Whether any of this holds upon Linux**, upon hardware without efficiency cores, or upon a Ruby which is not 4.0.5. Every figure here is one machine's, over two days.
|
data/README.md
ADDED
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
# farm.rb
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
## Description
|
|
5
|
+
|
|
6
|
+
Distribute work across CPUs by measuring what the machine affords, rather than by the caller naming a concurrency primitive.
|
|
7
|
+
|
|
8
|
+
The caller knows the shape of its work — map these rows, do this to each item — and nothing about ractors, threads, processes, or how many of them this machine can usefully hold. Which substrate to run upon and how wide to run it follow from measurement: the cores there are, the memory there is, how busy the machine already is, and what things of this kind have cost to run before. Naming a primitive in the caller fixes all of that at write time, upon whoever's laptop happened to be in front of the author.
|
|
9
|
+
|
|
10
|
+
This is a walking skeleton, 0.0.0, and most of that deciding is not here yet. What is here: the API that hides it, two executors which run it, the probe that will inform it, and the fall-back that makes a wrong guess safe — `Farm::RactorExecutor` runs the work across a pool of ractors as wide as the machine has performance cores, and anything which cannot cross the boundary, or that machines it, is retried serially rather than surfaced. The caller gets the right answer slowly instead of an exception quickly, and the library gets the observation.
|
|
11
|
+
|
|
12
|
+
Work travels as a name rather than a block: `Farm.map(rows, Revenue)` names the receiver and names `call`, or another method, upon it, so the work itself can cross to a worker. A block cannot be named over the boundary and blocks are run serially, faithfully and without pretence of parallelism.
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
## Installation
|
|
16
|
+
|
|
17
|
+
Add this line to your application's Gemfile:
|
|
18
|
+
|
|
19
|
+
```ruby
|
|
20
|
+
gem 'farm.rb'
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
And then execute:
|
|
24
|
+
|
|
25
|
+
```bash
|
|
26
|
+
$ bundle install
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
Or install it yourself as:
|
|
30
|
+
|
|
31
|
+
```bash
|
|
32
|
+
$ gem install farm.rb
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
## Usage
|
|
37
|
+
|
|
38
|
+
```ruby
|
|
39
|
+
require 'farm.rb'
|
|
40
|
+
|
|
41
|
+
module Revenue
|
|
42
|
+
def self.call(row)
|
|
43
|
+
row.quantity * row.price
|
|
44
|
+
end
|
|
45
|
+
|
|
46
|
+
def self.net(row)
|
|
47
|
+
call(row) - row.discount
|
|
48
|
+
end
|
|
49
|
+
end
|
|
50
|
+
|
|
51
|
+
Farm.map(rows, Revenue)
|
|
52
|
+
# => the map of Revenue.call over rows, spread across the performance cores
|
|
53
|
+
|
|
54
|
+
Farm.map(rows, Revenue, :net)
|
|
55
|
+
# => the same, naming the method
|
|
56
|
+
|
|
57
|
+
Farm.each(rows, Revenue)
|
|
58
|
+
# => run for effect; returns rows
|
|
59
|
+
|
|
60
|
+
Farm.map(rows){|row| row.upcase}
|
|
61
|
+
# => blocks cannot cross to a worker and are run serially
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
Whatever the executor cannot run it falls back from: work touching globals or shared state raises inside a worker, is tagged by the pool, and the whole run is retried serially in the caller's own process, where such work belongs. `Farm::Probe` exposes the measurements the deciding will rest upon — `Farm::Probe.performance_cores`, `logical_cores`, `total_memory`, `resident_memory`, `load_averages` — each asked of the platform and nil where the platform does not answer.
|
|
65
|
+
|
|
66
|
+
Ruby 4.0 or later, the Ractor API having changed at 4.0: `Ractor::Port` is the queue and `#value` answers where `#take` used to. Ruby still calls the API experimental and says so once per process upon the first worker spawned; the warning is Ruby's and is left alone.
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
## Contributing
|
|
70
|
+
|
|
71
|
+
1. Fork it ( https://github.com/thoran/farm/fork )
|
|
72
|
+
2. Create your feature branch (`git checkout -b my-new-feature`)
|
|
73
|
+
3. Commit your changes (`git commit -am 'Add some feature'`)
|
|
74
|
+
4. Push to the branch (`git push origin my-new-feature`)
|
|
75
|
+
5. Create a new pull request
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
## License
|
|
79
|
+
|
|
80
|
+
MIT
|
data/ROADMAP.md
ADDED
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
# Farm Roadmap
|
|
2
|
+
|
|
3
|
+
Date: 20261004
|
|
4
|
+
|
|
5
|
+
The figures quoted here were measured for the library and are recorded in MEASUREMENTS.md — with the machine, the method, and the run-to-run spread. They travel as priors to be corrected upon the machine in hand, not as constants.
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
## Design decisions
|
|
9
|
+
|
|
10
|
+
Recorded as 0.0.0 was cut. The version entries beneath may change; these hold.
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
### The work travels as a name, or it does not travel
|
|
14
|
+
|
|
15
|
+
A Proc cannot cross a Ractor boundary, and nothing launders one: `define_method` from a Proc keeps the unshareable body it was defined from — isolation is decided by the Proc, not by the method built around it — and a Method object cannot cross whatever it was built from. A Proc built at the top level cannot even be isolated in place, `self` there being `main`. What runs is a module reached by name.
|
|
16
|
+
|
|
17
|
+
Farm therefore asks for a callable and a method upon it, runs a block serially and says so, and leaves hosting to the consumer: render the body as source into a module of one's own and it crosses as well as anything typed by hand. The one route which should not be taken is regenerating a Proc's source into a shared carrier — a captured local silently rebinds to a same-named method already collected there, the answer changes, and nothing complains. Work should be born hosted: source first, closure never. (Measured, surveys and tables in MEASUREMENTS.md.)
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
### Embeddable, in both directions
|
|
21
|
+
|
|
22
|
+
Farm is meant to sit underneath other libraries, and to mean nothing to their users. The whole contract is `map(enumerable, callable, method)`: work travels as a name. A consumer renders whatever it has — Namo's formulae, or anything else's named computation — into a module and a method upon it, and Farm cannot tell that module from one typed by hand. Hosting source-carried computation is the consumer's business: eval and its trust questions never enter Farm.
|
|
23
|
+
|
|
24
|
+
Invisibility runs both ways. The host's users never meet Farm — the consumer soft-requires the gem, and without it everything runs serially and answers identically. Farm never meets the host's domain — no rows, no formulae, nothing above `map`'s three arguments. Farm.rb stays swappable for anything answering the same three-argument shape, and nothing upstream of the consumer needs to change.
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
### Observational equivalence
|
|
28
|
+
|
|
29
|
+
Every executor and every fall-back answers the same values in the same order, and raises the same exceptions serial would raise. Speed is the only admissible tell. This is the acceptance test for everything below and everything after: a parallel run which differs from the serial run in anything but time is a bug, wherever the difference surfaced.
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
### Measured, never tabulated
|
|
33
|
+
|
|
34
|
+
Width, substrate, and whether a job is worth spreading are read off the machine and the work in front of it, never configured by the caller and never assumed from a table. Four rules of practice, each bought with a measurement which first came out wrong:
|
|
35
|
+
|
|
36
|
+
- Whether isolation will allow the work is a property of the method body, not the payload; five of eight surveyed payloads passed every check at the boundary and failed on their first line of work. Static checks predict almost nothing, so Farm speculates and sees.
|
|
37
|
+
- `nproc` is the wrong width wherever efficiency cores exist; the performance-core count is the one that matters, and 0.0.0's `Probe.performance_cores` already answers it.
|
|
38
|
+
- `Ractor.make_shareable` is never called upon data Farm does not own. It deep-freezes whatever it reaches — globals included — and succeeds while doing it; an object holding `STDOUT` took `STDOUT` down with it process-wide, every later `puts` raising. `freeze` is shallow where shareability is deep, so a frozen constant is still refused from a Ractor. The advice to callers runs the other way: wrap your own constants at declaration, which repays itself within one dispatch.
|
|
39
|
+
- Every measured figure is a prior with a noise floor. Run-to-run spread for `fork` is around forty-five per cent, so a difference under about two-fold is no finding, and one run of anything will always name a winner. More than one conclusion here was first arrived at wrongly by an instrument returning a plausible figure for the wrong question; where a number underlies a decision it is re-measured rather than trusted.
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
## 0.1.0: Embeddable
|
|
43
|
+
|
|
44
|
+
The version which makes sitting underneath possible — the fall-back closed over transport, the substrate set widened beyond ractors, and the first measurement feeding the deciding.
|
|
45
|
+
|
|
46
|
+
1. The send-time rescue gap closes. A job which cannot be copied over the boundary — a Proc or a Method somewhere in its object graph — raises at the feed point, outside the worker-tag rescue, so 0.0.0's fall-back misses it and the caller meets a TypeError. Fold transport failure into NotParallelisable: whatever cannot be carried is run serially, exactly as whatever cannot be run. A small change, and the piece that makes "none the wiser" true.
|
|
47
|
+
|
|
48
|
+
2. + Farm::ForkExecutor: a pool of forked processes, the one substrate which imposes no isolation rules at all — the child holds the whole process, so blocks run as written and code touching globals or class state just works. Its costs are measured and they decide when it is used rather than whether it exists: about a millisecond to start against a Ractor's 0.033; a cost which belongs to the calling process rather than the machine, rising with the resident set's high-water mark — twenty-fold at a gigabyte held — and never coming back down; results home through Marshal and a pipe at about 380MB/s, so a large enough answer argues against fork on its own. Its startup figure is estimated from `Probe.resident_memory` times a calibrated rate and corrected from the first real dispatch — never probed at startup, the probe costing more than it settles. Fork is the fall-back for computation, not the default substrate.
|
|
49
|
+
|
|
50
|
+
3. + ThreadExecutor: the fall-back for waiting work. Threads do not scale upon computation — the GVL is held throughout, 1.00x at every width — and scale nearly perfectly upon waiting, where a worker occupies no core. So the measured policy stands: try a Ractor first whatever the work, and fall back by class — threads for waiting, fork for computation. Nothing recommends `spawn` for Ruby work; exec'ing the interpreter afresh costs forty milliseconds a unit.
|
|
51
|
+
|
|
52
|
+
4. The measurement loop goes in. Three inputs decide: the job's total serial work, the per-unit duration, and the size of what comes back. Whether to parallelise is decided upon the job, not the unit — a pool pays setup per worker rather than per unit, so `N·D > W·S/(1−1/k)` per candidate substrate and width: a Ractor pool of eight wants about a third of a millisecond of work, of forty-eight about 1.6ms, fork about 12ms in a small process. Which class the work is: a two-width sweep watching whether wall time falls as workers rise — a few milliseconds, and no understanding of the work required. Whether the work will cross: speculate one unit inside a Ractor, one unit plus 0.03ms, whereupon isolation refusal routes to the class's fall-back and anything else is the work's own error and re-raises. Below the threshold for every substrate the job runs serially; units too small to clear it are batched until the batch does.
|
|
53
|
+
|
|
54
|
+
5. + Wisdom: an in-process memo of observed cost against enumerable size, keyed by the callable, so the second run of the same shape of work starts informed. The memo is written to and read from through the noise-floor rule above — a figure recorded from one run is a hint, not a measurement. Persistence is a later question; the shape of the record is this one's.
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
## Not yet
|
|
58
|
+
|
|
59
|
+
- Distribution across machines: more executors, workers beyond the host, and the measurement to know when a socket beats a core — a same-chassis peer already costs less to reach than fork does. Substrates where measurement stays predictive (short-tailed latency, loud failure, un-withdrawable capacity, no price) may be selected among freely; the rest want the caller's explicit opt-in, under the standing principle that an autotuner may spend time without asking but never money without being told. The wisdom cache's persistence format belongs to this same body of work.
|
|
60
|
+
- Hybrid arrangements — ractors feeding forks, width varying with load: measured once, inconclusively. The single-substrate figures are the corners of the space, and the differences between arrangements sat beneath the differences between runs of the same arrangement. Any later search has to beat the noise floor before it beats the field; answering *whether a search has found anything at all* is the real work there.
|
|
61
|
+
- Fibers: unmeasured. Waiting at a scale a native thread apiece would not suit.
|
|
62
|
+
- Public tuning: width, substrate, and threshold are measured, not configured, and stay that way.
|
|
63
|
+
- Consumer-side hosting: Namo compiling its formulae into named modules and calling Farm from its batch-evaluation paths is a Namo release, soft-requiring farm.rb on top of whichever version shipped. Nothing about it lands here.
|
|
64
|
+
- Re-measurement: everything in MEASUREMENTS.md is one machine's, two days in August 2026, ruby 4.0.5. Re-measure upon new hardware, new platforms and new rubies before trusting a figure outside its spread; the methods are described there beside the numbers.
|
data/Rakefile
ADDED
data/farm.rb.gemspec
ADDED
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
require_relative './lib/Farm/VERSION'
|
|
2
|
+
|
|
3
|
+
class Gem::Specification
|
|
4
|
+
def dependencies=(gems)
|
|
5
|
+
gems.each{|gem| add_dependency(*gem)}
|
|
6
|
+
end
|
|
7
|
+
|
|
8
|
+
def development_dependencies=(gems)
|
|
9
|
+
gems.each{|gem| add_development_dependency(*gem)}
|
|
10
|
+
end
|
|
11
|
+
end
|
|
12
|
+
|
|
13
|
+
Gem::Specification.new do |spec|
|
|
14
|
+
spec.name = 'farm.rb'
|
|
15
|
+
spec.version = Farm::VERSION
|
|
16
|
+
|
|
17
|
+
spec.summary = "Parallelism decided by measurement."
|
|
18
|
+
spec.description = "Distribute work across CPUs by measuring what the machine affords, rather than by the caller naming a concurrency primitive."
|
|
19
|
+
|
|
20
|
+
spec.author = 'thoran'
|
|
21
|
+
spec.email = 'code@thoran.com'
|
|
22
|
+
spec.homepage = 'https://github.com/thoran/farm'
|
|
23
|
+
spec.license = 'MIT'
|
|
24
|
+
|
|
25
|
+
spec.require_paths = ['lib']
|
|
26
|
+
spec.required_ruby_version = '>= 4.0'
|
|
27
|
+
|
|
28
|
+
spec.files = [
|
|
29
|
+
'farm.rb.gemspec',
|
|
30
|
+
Dir['lib/**/*.rb'],
|
|
31
|
+
Dir['test/**/*.rb'],
|
|
32
|
+
'CHANGELOG',
|
|
33
|
+
'LICENSE',
|
|
34
|
+
'Gemfile',
|
|
35
|
+
'Rakefile',
|
|
36
|
+
'MEASUREMENTS.md',
|
|
37
|
+
'README.md',
|
|
38
|
+
'ROADMAP.md',
|
|
39
|
+
].flatten
|
|
40
|
+
|
|
41
|
+
spec.dependencies = []
|
|
42
|
+
|
|
43
|
+
spec.development_dependencies = %w{
|
|
44
|
+
minitest
|
|
45
|
+
minitest-spec-context
|
|
46
|
+
rake
|
|
47
|
+
}
|
|
48
|
+
end
|
data/lib/Farm/Probe.rb
ADDED
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
# Farm/Probe.rb
|
|
2
|
+
# Farm::Probe
|
|
3
|
+
|
|
4
|
+
require 'etc'
|
|
5
|
+
|
|
6
|
+
module Farm
|
|
7
|
+
# What the machine says about itself. Which substrate and how wide to run
|
|
8
|
+
# are decided from measurements rather than declared by the caller, so the
|
|
9
|
+
# measurements live here: the cores there are and which of them are fast,
|
|
10
|
+
# the memory there is and how much of it this process holds, and how busy
|
|
11
|
+
# the machine already is. Each figure is asked of the platform rather than
|
|
12
|
+
# tabulated, and platforms which do not answer get a nil rather than a
|
|
13
|
+
# guess — which a sandbox is a case of, sysctl and ps being politely
|
|
14
|
+
# refused there as much as they are absent off-POSIX, and the caller of a
|
|
15
|
+
# measurement being owed silence rather than the platform's error message.
|
|
16
|
+
module Probe
|
|
17
|
+
class << self
|
|
18
|
+
def logical_cores
|
|
19
|
+
Etc.nprocessors
|
|
20
|
+
end
|
|
21
|
+
|
|
22
|
+
# Darwin distinguishes the performance cores from the efficiency cores
|
|
23
|
+
# and reports the former under hw.perflevel0; latency-bound work belongs
|
|
24
|
+
# on them alone, so the pool width follows them where they are named.
|
|
25
|
+
def performance_cores
|
|
26
|
+
return logical_cores unless darwin?()
|
|
27
|
+
Integer(`sysctl -n hw.perflevel0.physicalcpu 2>/dev/null`.strip, exception: false) || logical_cores
|
|
28
|
+
end
|
|
29
|
+
|
|
30
|
+
def total_memory
|
|
31
|
+
return nil unless darwin?()
|
|
32
|
+
Integer(`sysctl -n hw.memsize 2>/dev/null`.strip, exception: false)
|
|
33
|
+
end
|
|
34
|
+
|
|
35
|
+
# ps reports rss in kilobytes upon both Darwin and Linux, so the figure
|
|
36
|
+
# is returned in bytes either way. Where ps may not be executed at all
|
|
37
|
+
# the spawn itself raises, and that refusal is nil too.
|
|
38
|
+
def resident_memory
|
|
39
|
+
if (kilobytes = Integer(`ps -o rss= -p #{Process.pid} 2>/dev/null`.strip, exception: false))
|
|
40
|
+
kilobytes * 1024
|
|
41
|
+
end
|
|
42
|
+
rescue SystemCallError
|
|
43
|
+
nil
|
|
44
|
+
end
|
|
45
|
+
|
|
46
|
+
def load_averages
|
|
47
|
+
if File.readable?('/proc/loadavg')
|
|
48
|
+
File.read('/proc/loadavg').split.first(3).map(&:to_f)
|
|
49
|
+
elsif darwin?()
|
|
50
|
+
averages = `sysctl -n vm.loadavg 2>/dev/null`.scan(/\d+\.\d+/).first(3).map(&:to_f)
|
|
51
|
+
averages unless averages.length < 3
|
|
52
|
+
end
|
|
53
|
+
end
|
|
54
|
+
|
|
55
|
+
def darwin?
|
|
56
|
+
RUBY_PLATFORM.include?('darwin')
|
|
57
|
+
end
|
|
58
|
+
end # class << self
|
|
59
|
+
end
|
|
60
|
+
end
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
# Farm/RactorExecutor.rb
|
|
2
|
+
# Farm::RactorExecutor
|
|
3
|
+
|
|
4
|
+
module Farm
|
|
5
|
+
# A pool of ractors fed round-robin, the results coming back over a single
|
|
6
|
+
# port and put back in order. Work which does not survive the crossing —
|
|
7
|
+
# globals, shared mutable state — is discovered inside the worker, tagged,
|
|
8
|
+
# and answered with NotParallelisable so the caller can run it serially;
|
|
9
|
+
# the Ractor API does its own discovering at spawn and send time, which
|
|
10
|
+
# arrives here as Ractor::Error and is met the same way by Farm. Workers
|
|
11
|
+
# rescue and tag rather than being allowed to die, a worker's death
|
|
12
|
+
# delivering nothing to the port and leaving its replacement to hang.
|
|
13
|
+
class RactorExecutor
|
|
14
|
+
class << self
|
|
15
|
+
def map(enumerable, callable, method = :call)
|
|
16
|
+
new(enumerable, callable, method).map
|
|
17
|
+
end
|
|
18
|
+
|
|
19
|
+
def each(enumerable, callable, method = :call)
|
|
20
|
+
new(enumerable, callable, method).each
|
|
21
|
+
end
|
|
22
|
+
end # class << self
|
|
23
|
+
|
|
24
|
+
def map
|
|
25
|
+
run
|
|
26
|
+
end
|
|
27
|
+
|
|
28
|
+
def each
|
|
29
|
+
run
|
|
30
|
+
@enumerable
|
|
31
|
+
end
|
|
32
|
+
|
|
33
|
+
private
|
|
34
|
+
|
|
35
|
+
def initialize(enumerable, callable, method)
|
|
36
|
+
@enumerable = enumerable
|
|
37
|
+
@jobs = enumerable.each_with_index.map{|item, index| [index, item]}
|
|
38
|
+
@callable = callable
|
|
39
|
+
@method = method
|
|
40
|
+
end
|
|
41
|
+
|
|
42
|
+
def width
|
|
43
|
+
[Probe.performance_cores, @jobs.length].min
|
|
44
|
+
end
|
|
45
|
+
|
|
46
|
+
def run
|
|
47
|
+
port = Ractor::Port.new
|
|
48
|
+
workers = width.times.map do
|
|
49
|
+
Ractor.new(port, @callable, @method) do |results, callable, method|
|
|
50
|
+
loop do
|
|
51
|
+
job = Ractor.receive
|
|
52
|
+
break if job.nil?
|
|
53
|
+
index, item = job
|
|
54
|
+
begin
|
|
55
|
+
results << [:ok, index, callable.public_send(method, item)]
|
|
56
|
+
rescue => error
|
|
57
|
+
results << [:error, index, error.class.name, error.message]
|
|
58
|
+
end
|
|
59
|
+
end
|
|
60
|
+
end
|
|
61
|
+
end
|
|
62
|
+
@jobs.each_with_index{|job, index| workers[index % width].send(job)}
|
|
63
|
+
workers.each{|worker| worker.send(nil)}
|
|
64
|
+
collected = Array.new(@jobs.length){port.receive}
|
|
65
|
+
workers.each(&:join)
|
|
66
|
+
collected.each do |entry|
|
|
67
|
+
if entry[0] == :error
|
|
68
|
+
_, _index, class_name, message = entry
|
|
69
|
+
raise NotParallelisable, "#{class_name} in a Ractor worker: #{message}"
|
|
70
|
+
end
|
|
71
|
+
end
|
|
72
|
+
collected.sort_by{|entry| entry[1]}.map{|entry| entry[2]}
|
|
73
|
+
end
|
|
74
|
+
end
|
|
75
|
+
end
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
# Farm/SerialExecutor.rb
|
|
2
|
+
# Farm::SerialExecutor
|
|
3
|
+
|
|
4
|
+
module Farm
|
|
5
|
+
# The ordinary way of doing the work, and what every parallel executor is
|
|
6
|
+
# measured against and falls back to. Being correct here is most of what
|
|
7
|
+
# being correct means: an executor which can disagree with the serial run
|
|
8
|
+
# does not get used.
|
|
9
|
+
class SerialExecutor
|
|
10
|
+
class << self
|
|
11
|
+
def map(enumerable, callable = nil, method = :call, &block)
|
|
12
|
+
if block
|
|
13
|
+
enumerable.map(&block)
|
|
14
|
+
else
|
|
15
|
+
enumerable.map{|item| callable.public_send(method, item)}
|
|
16
|
+
end
|
|
17
|
+
end
|
|
18
|
+
|
|
19
|
+
def each(enumerable, callable = nil, method = :call, &block)
|
|
20
|
+
if block
|
|
21
|
+
enumerable.each(&block)
|
|
22
|
+
else
|
|
23
|
+
enumerable.each{|item| callable.public_send(method, item)}
|
|
24
|
+
end
|
|
25
|
+
end
|
|
26
|
+
end # class << self
|
|
27
|
+
end
|
|
28
|
+
end
|
data/lib/Farm/VERSION.rb
ADDED
data/lib/farm.rb
ADDED
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
# farm.rb
|
|
2
|
+
# Farm
|
|
3
|
+
|
|
4
|
+
require_relative './Farm/NotParallelisable'
|
|
5
|
+
require_relative './Farm/Probe'
|
|
6
|
+
require_relative './Farm/SerialExecutor'
|
|
7
|
+
require_relative './Farm/RactorExecutor'
|
|
8
|
+
require_relative './Farm/VERSION'
|
|
9
|
+
|
|
10
|
+
module Farm
|
|
11
|
+
# Work travels as a name rather than a block: Farm.map(rows, Revenue) names
|
|
12
|
+
# the receiver and names it to a worker, where a block could not cross.
|
|
13
|
+
# Blocks are run serially, there being nothing to name. A RactorExecutor
|
|
14
|
+
# run which fails for the machinery's own reasons — the work touching what
|
|
15
|
+
# workers may not touch, the pool falling over — is retried serially rather
|
|
16
|
+
# than surfaced, the library having promised to run the work, not to run it
|
|
17
|
+
# in parallel.
|
|
18
|
+
class << self
|
|
19
|
+
def map(enumerable, callable = nil, method = :call, &block)
|
|
20
|
+
if block
|
|
21
|
+
SerialExecutor.map(enumerable, &block)
|
|
22
|
+
else
|
|
23
|
+
begin
|
|
24
|
+
RactorExecutor.map(enumerable, callable, method)
|
|
25
|
+
rescue NotParallelisable, Ractor::Error
|
|
26
|
+
SerialExecutor.map(enumerable, callable, method)
|
|
27
|
+
end
|
|
28
|
+
end
|
|
29
|
+
end
|
|
30
|
+
|
|
31
|
+
def each(enumerable, callable = nil, method = :call, &block)
|
|
32
|
+
if block
|
|
33
|
+
SerialExecutor.each(enumerable, &block)
|
|
34
|
+
else
|
|
35
|
+
begin
|
|
36
|
+
RactorExecutor.each(enumerable, callable, method)
|
|
37
|
+
rescue NotParallelisable, Ractor::Error
|
|
38
|
+
SerialExecutor.each(enumerable, callable, method)
|
|
39
|
+
end
|
|
40
|
+
end
|
|
41
|
+
end
|
|
42
|
+
end # class << self
|
|
43
|
+
end
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
require_relative '../../lib/farm.rb'
|
|
2
|
+
|
|
3
|
+
require 'minitest/autorun'
|
|
4
|
+
require 'minitest-spec-context'
|
|
5
|
+
|
|
6
|
+
describe Farm::Probe do
|
|
7
|
+
describe ".logical_cores" do
|
|
8
|
+
it "answers at least one" do
|
|
9
|
+
_(Farm::Probe.logical_cores).must_be :>=, 1
|
|
10
|
+
end
|
|
11
|
+
end
|
|
12
|
+
|
|
13
|
+
describe ".performance_cores" do
|
|
14
|
+
it "answers at least one, being the width the pool draws itself to" do
|
|
15
|
+
_(Farm::Probe.performance_cores).must_be :>=, 1
|
|
16
|
+
end
|
|
17
|
+
|
|
18
|
+
it "answers no more than the logical cores" do
|
|
19
|
+
_(Farm::Probe.performance_cores).must_be :<=, Farm::Probe.logical_cores
|
|
20
|
+
end
|
|
21
|
+
end
|
|
22
|
+
|
|
23
|
+
# The probe's contract being nil where the platform does not answer, these
|
|
24
|
+
# assert the shape of an answer without requiring the platform to answer:
|
|
25
|
+
# a sandboxed sysctl or ps is precisely a platform that does not.
|
|
26
|
+
describe ".total_memory" do
|
|
27
|
+
it "answers the machine's memory in bytes, where the platform answers at all" do
|
|
28
|
+
if (total_memory = Farm::Probe.total_memory)
|
|
29
|
+
_(total_memory).must_be_instance_of Integer
|
|
30
|
+
_(total_memory).must_be :>, 0
|
|
31
|
+
end
|
|
32
|
+
end
|
|
33
|
+
|
|
34
|
+
it "answers nothing off Darwin" do
|
|
35
|
+
_(Farm::Probe.total_memory).must_be_nil unless Farm::Probe.darwin?
|
|
36
|
+
end
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
describe ".resident_memory" do
|
|
40
|
+
it "answers this process's resident set in bytes, where ps answers at all" do
|
|
41
|
+
if (resident_memory = Farm::Probe.resident_memory)
|
|
42
|
+
_(resident_memory).must_be_instance_of Integer
|
|
43
|
+
_(resident_memory).must_be :>, 0
|
|
44
|
+
end
|
|
45
|
+
end
|
|
46
|
+
end
|
|
47
|
+
|
|
48
|
+
describe ".load_averages" do
|
|
49
|
+
it "answers three averages, where the platform answers at all" do
|
|
50
|
+
if (load_averages = Farm::Probe.load_averages)
|
|
51
|
+
_(load_averages.length).must_equal(3)
|
|
52
|
+
_(load_averages.all?{|average| average >= 0}).must_equal(true)
|
|
53
|
+
end
|
|
54
|
+
end
|
|
55
|
+
end
|
|
56
|
+
end
|
|
@@ -0,0 +1,62 @@
|
|
|
1
|
+
require_relative '../../lib/farm.rb'
|
|
2
|
+
|
|
3
|
+
require 'minitest/autorun'
|
|
4
|
+
require 'minitest-spec-context'
|
|
5
|
+
|
|
6
|
+
module FarmRactorExecutorTestDoubler
|
|
7
|
+
def self.call(item)
|
|
8
|
+
item * 2
|
|
9
|
+
end
|
|
10
|
+
|
|
11
|
+
def self.triple(item)
|
|
12
|
+
item * 3
|
|
13
|
+
end
|
|
14
|
+
|
|
15
|
+
# Results deliberately arriving out of order, so the executor's own
|
|
16
|
+
# re-ordering is the only thing which could put them back.
|
|
17
|
+
def self.slow_reversed(item)
|
|
18
|
+
sleep(0.001 * ((item * 7) % 11))
|
|
19
|
+
item
|
|
20
|
+
end
|
|
21
|
+
end
|
|
22
|
+
|
|
23
|
+
module FarmRactorExecutorTestBoom
|
|
24
|
+
def self.call(item)
|
|
25
|
+
$farm_ractor_executor_test_global = item
|
|
26
|
+
end
|
|
27
|
+
end
|
|
28
|
+
|
|
29
|
+
describe Farm::RactorExecutor do
|
|
30
|
+
describe ".map" do
|
|
31
|
+
it "maps with a callable, #call being the default method" do
|
|
32
|
+
_(Farm::RactorExecutor.map([1, 2, 3], FarmRactorExecutorTestDoubler)).must_equal([2, 4, 6])
|
|
33
|
+
end
|
|
34
|
+
|
|
35
|
+
it "maps with a callable and a named method" do
|
|
36
|
+
_(Farm::RactorExecutor.map([1, 2, 3], FarmRactorExecutorTestDoubler, :triple)).must_equal([3, 6, 9])
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
it "puts out-of-order results back in order" do
|
|
40
|
+
items = (1..40).to_a
|
|
41
|
+
_(Farm::RactorExecutor.map(items, FarmRactorExecutorTestDoubler, :slow_reversed)).must_equal(items)
|
|
42
|
+
end
|
|
43
|
+
|
|
44
|
+
it "runs an empty enumerable without a worker" do
|
|
45
|
+
_(Farm::RactorExecutor.map([], FarmRactorExecutorTestDoubler)).must_equal([])
|
|
46
|
+
end
|
|
47
|
+
|
|
48
|
+
it "answers NotParallelisable when the work cannot cross" do
|
|
49
|
+
error = _(proc{
|
|
50
|
+
Farm::RactorExecutor.map([1, 2, 3], FarmRactorExecutorTestBoom)
|
|
51
|
+
}).must_raise(Farm::NotParallelisable)
|
|
52
|
+
_(error.message).must_match(/Ractor::IsolationError/)
|
|
53
|
+
end
|
|
54
|
+
end
|
|
55
|
+
|
|
56
|
+
describe ".each" do
|
|
57
|
+
it "returns the enumerable" do
|
|
58
|
+
enumerable = [1, 2, 3]
|
|
59
|
+
_(Farm::RactorExecutor.each(enumerable, FarmRactorExecutorTestDoubler)).must_be_same_as(enumerable)
|
|
60
|
+
end
|
|
61
|
+
end
|
|
62
|
+
end
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
require_relative '../../lib/farm.rb'
|
|
2
|
+
|
|
3
|
+
require 'minitest/autorun'
|
|
4
|
+
require 'minitest-spec-context'
|
|
5
|
+
|
|
6
|
+
module FarmSerialExecutorTestDoubler
|
|
7
|
+
def self.call(item)
|
|
8
|
+
item * 2
|
|
9
|
+
end
|
|
10
|
+
|
|
11
|
+
def self.triple(item)
|
|
12
|
+
item * 3
|
|
13
|
+
end
|
|
14
|
+
end
|
|
15
|
+
|
|
16
|
+
describe Farm::SerialExecutor do
|
|
17
|
+
describe ".map" do
|
|
18
|
+
it "maps with a callable, #call being the default method" do
|
|
19
|
+
_(Farm::SerialExecutor.map([1, 2, 3], FarmSerialExecutorTestDoubler)).must_equal([2, 4, 6])
|
|
20
|
+
end
|
|
21
|
+
|
|
22
|
+
it "maps with a callable and a named method" do
|
|
23
|
+
_(Farm::SerialExecutor.map([1, 2, 3], FarmSerialExecutorTestDoubler, :triple)).must_equal([3, 6, 9])
|
|
24
|
+
end
|
|
25
|
+
|
|
26
|
+
it "maps with a block" do
|
|
27
|
+
_(Farm::SerialExecutor.map([1, 2, 3]){|item| item * 4}).must_equal([4, 8, 12])
|
|
28
|
+
end
|
|
29
|
+
end
|
|
30
|
+
|
|
31
|
+
describe ".each" do
|
|
32
|
+
it "runs the callable for each item and returns the enumerable" do
|
|
33
|
+
enumerable = [1, 2, 3]
|
|
34
|
+
_(Farm::SerialExecutor.each(enumerable, FarmSerialExecutorTestDoubler)).must_be_same_as(enumerable)
|
|
35
|
+
end
|
|
36
|
+
|
|
37
|
+
it "runs the block for each item, side effects included, being serial" do
|
|
38
|
+
seen = []
|
|
39
|
+
Farm::SerialExecutor.each([1, 2, 3]){|item| seen << item}
|
|
40
|
+
_(seen).must_equal([1, 2, 3])
|
|
41
|
+
end
|
|
42
|
+
end
|
|
43
|
+
end
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
require_relative '../../lib/farm.rb'
|
|
2
|
+
|
|
3
|
+
require 'minitest/autorun'
|
|
4
|
+
require 'minitest-spec-context'
|
|
5
|
+
|
|
6
|
+
describe Farm do
|
|
7
|
+
describe "VERSION" do
|
|
8
|
+
it "is a string" do
|
|
9
|
+
_(Farm::VERSION).must_be_instance_of String
|
|
10
|
+
end
|
|
11
|
+
|
|
12
|
+
it "is three numbers separated by dots" do
|
|
13
|
+
_(Farm::VERSION).must_match(/\A\d+\.\d+\.\d+\z/)
|
|
14
|
+
end
|
|
15
|
+
|
|
16
|
+
it "matches the newest entry in the CHANGELOG" do
|
|
17
|
+
changelog = File.read(File.expand_path('../../CHANGELOG', __dir__))
|
|
18
|
+
_(changelog[/^(\d+\.\d+\.\d+):/, 1]).must_equal Farm::VERSION
|
|
19
|
+
end
|
|
20
|
+
end
|
|
21
|
+
end
|
data/test/farm_test.rb
ADDED
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
require_relative '../lib/farm.rb'
|
|
2
|
+
|
|
3
|
+
require 'minitest/autorun'
|
|
4
|
+
require 'minitest-spec-context'
|
|
5
|
+
|
|
6
|
+
module FarmAPITestDoubler
|
|
7
|
+
def self.call(item)
|
|
8
|
+
item * 2
|
|
9
|
+
end
|
|
10
|
+
|
|
11
|
+
def self.triple(item)
|
|
12
|
+
item * 3
|
|
13
|
+
end
|
|
14
|
+
end
|
|
15
|
+
|
|
16
|
+
# Work which cannot cross a Ractor boundary, Farm's answer being to run it
|
|
17
|
+
# serially and in this process, where the side effect is then observable.
|
|
18
|
+
module FarmAPITestGlobalToucher
|
|
19
|
+
def self.call(item)
|
|
20
|
+
$farm_api_test_counter = ($farm_api_test_counter || 0) + 1
|
|
21
|
+
item * 2
|
|
22
|
+
end
|
|
23
|
+
end
|
|
24
|
+
|
|
25
|
+
describe Farm do
|
|
26
|
+
describe ".map" do
|
|
27
|
+
it "maps with a callable across the Ractor executor" do
|
|
28
|
+
_(Farm.map([1, 2, 3], FarmAPITestDoubler)).must_equal([2, 4, 6])
|
|
29
|
+
end
|
|
30
|
+
|
|
31
|
+
it "maps with a callable and a named method" do
|
|
32
|
+
_(Farm.map([1, 2, 3], FarmAPITestDoubler, :triple)).must_equal([3, 6, 9])
|
|
33
|
+
end
|
|
34
|
+
|
|
35
|
+
it "maps a block serially, side effects landing in this process" do
|
|
36
|
+
seen = []
|
|
37
|
+
result = Farm.map([1, 2, 3]){|item| seen << item; item * 2}
|
|
38
|
+
_(result).must_equal([2, 4, 6])
|
|
39
|
+
_(seen).must_equal([1, 2, 3])
|
|
40
|
+
end
|
|
41
|
+
|
|
42
|
+
it "falls back to the serial executor when the work cannot cross" do
|
|
43
|
+
$farm_api_test_counter = 0
|
|
44
|
+
_(Farm.map([1, 2, 3], FarmAPITestGlobalToucher)).must_equal([2, 4, 6])
|
|
45
|
+
_($farm_api_test_counter).must_equal(3)
|
|
46
|
+
end
|
|
47
|
+
end
|
|
48
|
+
|
|
49
|
+
describe ".each" do
|
|
50
|
+
it "runs the callable and returns the enumerable" do
|
|
51
|
+
enumerable = [1, 2, 3]
|
|
52
|
+
_(Farm.each(enumerable, FarmAPITestDoubler)).must_be_same_as(enumerable)
|
|
53
|
+
end
|
|
54
|
+
|
|
55
|
+
it "runs a block serially, side effects landing in this process" do
|
|
56
|
+
seen = []
|
|
57
|
+
Farm.each([1, 2, 3]){|item| seen << item * 2}
|
|
58
|
+
_(seen).must_equal([2, 4, 6])
|
|
59
|
+
end
|
|
60
|
+
|
|
61
|
+
it "falls back to the serial executor when the work cannot cross" do
|
|
62
|
+
$farm_api_test_counter = 0
|
|
63
|
+
Farm.each([1, 2, 3], FarmAPITestGlobalToucher)
|
|
64
|
+
_($farm_api_test_counter).must_equal(3)
|
|
65
|
+
end
|
|
66
|
+
end
|
|
67
|
+
end
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
require_relative '../lib/farm.rb'
|
|
2
|
+
|
|
3
|
+
require 'minitest/autorun'
|
|
4
|
+
require 'minitest-spec-context'
|
|
5
|
+
|
|
6
|
+
describe 'farm.rb.gemspec' do
|
|
7
|
+
let(:spec){Gem::Specification.load(File.expand_path('../farm.rb.gemspec', __dir__))}
|
|
8
|
+
|
|
9
|
+
it "is a valid specification" do
|
|
10
|
+
_(spec.validate).must_equal(true)
|
|
11
|
+
end
|
|
12
|
+
|
|
13
|
+
it "does not pin a date" do
|
|
14
|
+
_(spec.date).must_equal(Gem::Specification.new.date)
|
|
15
|
+
end
|
|
16
|
+
|
|
17
|
+
it "takes its version from Farm::VERSION" do
|
|
18
|
+
_(spec.version.to_s).must_equal(Farm::VERSION)
|
|
19
|
+
end
|
|
20
|
+
|
|
21
|
+
it "declares no runtime dependencies" do
|
|
22
|
+
_(spec.runtime_dependencies).must_equal([])
|
|
23
|
+
end
|
|
24
|
+
|
|
25
|
+
it "declares its development dependencies" do
|
|
26
|
+
_(spec.development_dependencies.map(&:name).sort).must_equal(%w{minitest minitest-spec-context rake})
|
|
27
|
+
end
|
|
28
|
+
|
|
29
|
+
it "asks for the Ruby whose Ractor API it was written against" do
|
|
30
|
+
expect(Gem::Requirement.new('>= 4.0').satisfied_by?(Gem::Version.new(RUBY_VERSION))).must_equal(true)
|
|
31
|
+
_(spec.required_ruby_version.to_s).must_equal('>= 4.0')
|
|
32
|
+
end
|
|
33
|
+
end
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
require_relative '../lib/farm.rb'
|
|
2
|
+
|
|
3
|
+
require 'minitest/autorun'
|
|
4
|
+
require 'minitest-spec-context'
|
|
5
|
+
|
|
6
|
+
# In its own process, since this one has loaded every file directly and so could
|
|
7
|
+
# not tell what requiring the gem alone would bring.
|
|
8
|
+
describe 'loading' do
|
|
9
|
+
def ruby(source)
|
|
10
|
+
lib = File.expand_path('../lib', __dir__)
|
|
11
|
+
IO.popen(['ruby', "-I#{lib}", '-e', source], err: [:child, :out]){|io| io.read}.strip
|
|
12
|
+
end
|
|
13
|
+
|
|
14
|
+
it "defines the public API upon requiring the gem" do
|
|
15
|
+
expect(ruby('require "farm.rb"; print Farm.respond_to?(:map) && Farm.respond_to?(:each)')) \
|
|
16
|
+
.must_equal('true')
|
|
17
|
+
end
|
|
18
|
+
|
|
19
|
+
it "defines the executors, probe and failure class" do
|
|
20
|
+
expect(ruby('require "farm.rb"; print [defined?(Farm::SerialExecutor), defined?(Farm::RactorExecutor), defined?(Farm::Probe), defined?(Farm::NotParallelisable)].inspect')) \
|
|
21
|
+
.must_equal('["constant", "constant", "constant", "constant"]')
|
|
22
|
+
end
|
|
23
|
+
|
|
24
|
+
it "defines the version too" do
|
|
25
|
+
expect(ruby('require "farm.rb"; print Farm::VERSION')) \
|
|
26
|
+
.must_equal(Farm::VERSION)
|
|
27
|
+
end
|
|
28
|
+
end
|
metadata
ADDED
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
--- !ruby/object:Gem::Specification
|
|
2
|
+
name: farm.rb
|
|
3
|
+
version: !ruby/object:Gem::Version
|
|
4
|
+
version: 0.0.0
|
|
5
|
+
platform: ruby
|
|
6
|
+
authors:
|
|
7
|
+
- thoran
|
|
8
|
+
bindir: bin
|
|
9
|
+
cert_chain: []
|
|
10
|
+
date: 1980-01-02 00:00:00.000000000 Z
|
|
11
|
+
dependencies:
|
|
12
|
+
- !ruby/object:Gem::Dependency
|
|
13
|
+
name: minitest
|
|
14
|
+
requirement: !ruby/object:Gem::Requirement
|
|
15
|
+
requirements:
|
|
16
|
+
- - ">="
|
|
17
|
+
- !ruby/object:Gem::Version
|
|
18
|
+
version: '0'
|
|
19
|
+
type: :development
|
|
20
|
+
prerelease: false
|
|
21
|
+
version_requirements: !ruby/object:Gem::Requirement
|
|
22
|
+
requirements:
|
|
23
|
+
- - ">="
|
|
24
|
+
- !ruby/object:Gem::Version
|
|
25
|
+
version: '0'
|
|
26
|
+
- !ruby/object:Gem::Dependency
|
|
27
|
+
name: minitest-spec-context
|
|
28
|
+
requirement: !ruby/object:Gem::Requirement
|
|
29
|
+
requirements:
|
|
30
|
+
- - ">="
|
|
31
|
+
- !ruby/object:Gem::Version
|
|
32
|
+
version: '0'
|
|
33
|
+
type: :development
|
|
34
|
+
prerelease: false
|
|
35
|
+
version_requirements: !ruby/object:Gem::Requirement
|
|
36
|
+
requirements:
|
|
37
|
+
- - ">="
|
|
38
|
+
- !ruby/object:Gem::Version
|
|
39
|
+
version: '0'
|
|
40
|
+
- !ruby/object:Gem::Dependency
|
|
41
|
+
name: rake
|
|
42
|
+
requirement: !ruby/object:Gem::Requirement
|
|
43
|
+
requirements:
|
|
44
|
+
- - ">="
|
|
45
|
+
- !ruby/object:Gem::Version
|
|
46
|
+
version: '0'
|
|
47
|
+
type: :development
|
|
48
|
+
prerelease: false
|
|
49
|
+
version_requirements: !ruby/object:Gem::Requirement
|
|
50
|
+
requirements:
|
|
51
|
+
- - ">="
|
|
52
|
+
- !ruby/object:Gem::Version
|
|
53
|
+
version: '0'
|
|
54
|
+
description: Distribute work across CPUs by measuring what the machine affords, rather
|
|
55
|
+
than by the caller naming a concurrency primitive.
|
|
56
|
+
email: code@thoran.com
|
|
57
|
+
executables: []
|
|
58
|
+
extensions: []
|
|
59
|
+
extra_rdoc_files: []
|
|
60
|
+
files:
|
|
61
|
+
- CHANGELOG
|
|
62
|
+
- Gemfile
|
|
63
|
+
- LICENSE
|
|
64
|
+
- MEASUREMENTS.md
|
|
65
|
+
- README.md
|
|
66
|
+
- ROADMAP.md
|
|
67
|
+
- Rakefile
|
|
68
|
+
- farm.rb.gemspec
|
|
69
|
+
- lib/Farm/NotParallelisable.rb
|
|
70
|
+
- lib/Farm/Probe.rb
|
|
71
|
+
- lib/Farm/RactorExecutor.rb
|
|
72
|
+
- lib/Farm/SerialExecutor.rb
|
|
73
|
+
- lib/Farm/VERSION.rb
|
|
74
|
+
- lib/farm.rb
|
|
75
|
+
- test/Farm/Probe_test.rb
|
|
76
|
+
- test/Farm/RactorExecutor_test.rb
|
|
77
|
+
- test/Farm/SerialExecutor_test.rb
|
|
78
|
+
- test/Farm/VERSION_test.rb
|
|
79
|
+
- test/farm_test.rb
|
|
80
|
+
- test/gemspec_test.rb
|
|
81
|
+
- test/loading_test.rb
|
|
82
|
+
homepage: https://github.com/thoran/farm
|
|
83
|
+
licenses:
|
|
84
|
+
- MIT
|
|
85
|
+
metadata: {}
|
|
86
|
+
rdoc_options: []
|
|
87
|
+
require_paths:
|
|
88
|
+
- lib
|
|
89
|
+
required_ruby_version: !ruby/object:Gem::Requirement
|
|
90
|
+
requirements:
|
|
91
|
+
- - ">="
|
|
92
|
+
- !ruby/object:Gem::Version
|
|
93
|
+
version: '4.0'
|
|
94
|
+
required_rubygems_version: !ruby/object:Gem::Requirement
|
|
95
|
+
requirements:
|
|
96
|
+
- - ">="
|
|
97
|
+
- !ruby/object:Gem::Version
|
|
98
|
+
version: '0'
|
|
99
|
+
requirements: []
|
|
100
|
+
rubygems_version: 4.0.21
|
|
101
|
+
specification_version: 4
|
|
102
|
+
summary: Parallelism decided by measurement.
|
|
103
|
+
test_files: []
|