tigerbeetle-node 0.11.8 → 0.11.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (85) hide show
  1. package/dist/.client.node.sha256 +1 -1
  2. package/package.json +4 -3
  3. package/scripts/build_lib.sh +41 -0
  4. package/src/node.zig +1 -1
  5. package/src/tigerbeetle/scripts/validate_docs.sh +7 -1
  6. package/src/tigerbeetle/src/benchmark.zig +3 -3
  7. package/src/tigerbeetle/src/config.zig +31 -16
  8. package/src/tigerbeetle/src/constants.zig +48 -9
  9. package/src/tigerbeetle/src/ewah.zig +5 -5
  10. package/src/tigerbeetle/src/ewah_fuzz.zig +1 -1
  11. package/src/tigerbeetle/src/lsm/binary_search.zig +1 -1
  12. package/src/tigerbeetle/src/lsm/bloom_filter.zig +1 -1
  13. package/src/tigerbeetle/src/lsm/compaction.zig +34 -21
  14. package/src/tigerbeetle/src/lsm/forest_fuzz.zig +84 -104
  15. package/src/tigerbeetle/src/lsm/grid.zig +19 -13
  16. package/src/tigerbeetle/src/lsm/manifest_log.zig +8 -10
  17. package/src/tigerbeetle/src/lsm/manifest_log_fuzz.zig +18 -13
  18. package/src/tigerbeetle/src/lsm/merge_iterator.zig +1 -1
  19. package/src/tigerbeetle/src/lsm/segmented_array.zig +17 -17
  20. package/src/tigerbeetle/src/lsm/segmented_array_fuzz.zig +1 -1
  21. package/src/tigerbeetle/src/lsm/set_associative_cache.zig +1 -1
  22. package/src/tigerbeetle/src/lsm/table.zig +8 -20
  23. package/src/tigerbeetle/src/lsm/table_immutable.zig +1 -1
  24. package/src/tigerbeetle/src/lsm/table_iterator.zig +3 -3
  25. package/src/tigerbeetle/src/lsm/table_mutable.zig +14 -2
  26. package/src/tigerbeetle/src/lsm/test.zig +5 -4
  27. package/src/tigerbeetle/src/lsm/tree.zig +1 -2
  28. package/src/tigerbeetle/src/lsm/tree_fuzz.zig +85 -115
  29. package/src/tigerbeetle/src/message_bus.zig +4 -4
  30. package/src/tigerbeetle/src/message_pool.zig +7 -10
  31. package/src/tigerbeetle/src/ring_buffer.zig +22 -12
  32. package/src/tigerbeetle/src/simulator.zig +366 -239
  33. package/src/tigerbeetle/src/state_machine/auditor.zig +5 -5
  34. package/src/tigerbeetle/src/state_machine/workload.zig +3 -3
  35. package/src/tigerbeetle/src/state_machine.zig +190 -178
  36. package/src/tigerbeetle/src/{util.zig → stdx.zig} +2 -0
  37. package/src/tigerbeetle/src/storage.zig +13 -6
  38. package/src/tigerbeetle/src/{test → testing/cluster}/message_bus.zig +3 -3
  39. package/src/tigerbeetle/src/{test → testing/cluster}/network.zig +46 -22
  40. package/src/tigerbeetle/src/testing/cluster/state_checker.zig +169 -0
  41. package/src/tigerbeetle/src/testing/cluster/storage_checker.zig +202 -0
  42. package/src/tigerbeetle/src/testing/cluster.zig +443 -0
  43. package/src/tigerbeetle/src/{test → testing}/fuzz.zig +0 -0
  44. package/src/tigerbeetle/src/testing/hash_log.zig +66 -0
  45. package/src/tigerbeetle/src/{test → testing}/id.zig +0 -0
  46. package/src/tigerbeetle/src/testing/packet_simulator.zig +365 -0
  47. package/src/tigerbeetle/src/{test → testing}/priority_queue.zig +1 -1
  48. package/src/tigerbeetle/src/testing/reply_sequence.zig +139 -0
  49. package/src/tigerbeetle/src/{test → testing}/state_machine.zig +3 -1
  50. package/src/tigerbeetle/src/testing/storage.zig +757 -0
  51. package/src/tigerbeetle/src/{test → testing}/table.zig +21 -0
  52. package/src/tigerbeetle/src/{test → testing}/time.zig +0 -0
  53. package/src/tigerbeetle/src/tigerbeetle.zig +2 -0
  54. package/src/tigerbeetle/src/tracer.zig +3 -3
  55. package/src/tigerbeetle/src/unit_tests.zig +4 -4
  56. package/src/tigerbeetle/src/vopr.zig +2 -2
  57. package/src/tigerbeetle/src/vsr/client.zig +5 -2
  58. package/src/tigerbeetle/src/vsr/clock.zig +93 -53
  59. package/src/tigerbeetle/src/vsr/journal.zig +109 -98
  60. package/src/tigerbeetle/src/vsr/journal_format_fuzz.zig +2 -2
  61. package/src/tigerbeetle/src/vsr/replica.zig +1983 -1430
  62. package/src/tigerbeetle/src/vsr/replica_format.zig +13 -13
  63. package/src/tigerbeetle/src/vsr/superblock.zig +240 -142
  64. package/src/tigerbeetle/src/vsr/superblock_client_table.zig +7 -7
  65. package/src/tigerbeetle/src/vsr/superblock_free_set.zig +1 -1
  66. package/src/tigerbeetle/src/vsr/superblock_free_set_fuzz.zig +1 -1
  67. package/src/tigerbeetle/src/vsr/superblock_fuzz.zig +49 -14
  68. package/src/tigerbeetle/src/vsr/superblock_manifest.zig +38 -19
  69. package/src/tigerbeetle/src/vsr/superblock_quorums.zig +48 -48
  70. package/src/tigerbeetle/src/vsr/superblock_quorums_fuzz.zig +51 -51
  71. package/src/tigerbeetle/src/vsr.zig +99 -33
  72. package/src/tigerbeetle/src/demo.zig +0 -132
  73. package/src/tigerbeetle/src/demo_01_create_accounts.zig +0 -35
  74. package/src/tigerbeetle/src/demo_02_lookup_accounts.zig +0 -7
  75. package/src/tigerbeetle/src/demo_03_create_transfers.zig +0 -37
  76. package/src/tigerbeetle/src/demo_04_create_pending_transfers.zig +0 -61
  77. package/src/tigerbeetle/src/demo_05_post_pending_transfers.zig +0 -37
  78. package/src/tigerbeetle/src/demo_06_void_pending_transfers.zig +0 -24
  79. package/src/tigerbeetle/src/demo_07_lookup_transfers.zig +0 -7
  80. package/src/tigerbeetle/src/test/cluster.zig +0 -352
  81. package/src/tigerbeetle/src/test/conductor.zig +0 -366
  82. package/src/tigerbeetle/src/test/packet_simulator.zig +0 -398
  83. package/src/tigerbeetle/src/test/state_checker.zig +0 -169
  84. package/src/tigerbeetle/src/test/storage.zig +0 -864
  85. package/src/tigerbeetle/src/test/storage_checker.zig +0 -204
@@ -0,0 +1,443 @@
1
+ const std = @import("std");
2
+ const assert = std.debug.assert;
3
+ const mem = std.mem;
4
+
5
+ const constants = @import("../constants.zig");
6
+ const message_pool = @import("../message_pool.zig");
7
+ const MessagePool = message_pool.MessagePool;
8
+ const Message = MessagePool.Message;
9
+
10
+ const Storage = @import("storage.zig").Storage;
11
+ const StorageFaultAtlas = @import("storage.zig").ClusterFaultAtlas;
12
+ const Time = @import("time.zig").Time;
13
+ const IdPermutation = @import("id.zig").IdPermutation;
14
+
15
+ const MessageBus = @import("cluster/message_bus.zig").MessageBus;
16
+ const Network = @import("cluster/network.zig").Network;
17
+ const NetworkOptions = @import("cluster/network.zig").NetworkOptions;
18
+ const StateCheckerType = @import("cluster/state_checker.zig").StateCheckerType;
19
+ const StorageCheckerType = @import("cluster/storage_checker.zig").StorageCheckerType;
20
+
21
+ const vsr = @import("../vsr.zig");
22
+ pub const ReplicaFormat = vsr.ReplicaFormatType(Storage);
23
+ const SuperBlock = vsr.SuperBlockType(Storage);
24
+ const superblock_zone_size = @import("../vsr/superblock.zig").superblock_zone_size;
25
+
26
+ pub const ReplicaHealth = enum { up, down };
27
+
28
+ /// Integer values represent exit codes.
29
+ // TODO This doesn't really belong in Cluster, but it is needed here so that StateChecker failures
30
+ // use the particular exit code.
31
+ pub const Failure = enum(u8) {
32
+ /// Any assertion crash will be given an exit code of 127 by default.
33
+ crash = 127,
34
+ liveness = 128,
35
+ correctness = 129,
36
+ };
37
+
38
+ /// Shift the id-generating index because the simulator network expects client ids to never collide
39
+ /// with a replica index.
40
+ const client_id_permutation_shift = constants.replicas_max;
41
+
42
+ pub fn ClusterType(comptime StateMachineType: fn (comptime Storage: type, comptime constants: anytype) type) type {
43
+ return struct {
44
+ const Self = @This();
45
+
46
+ pub const StateMachine = StateMachineType(Storage, .{
47
+ .message_body_size_max = constants.message_body_size_max,
48
+ });
49
+ pub const Replica = vsr.ReplicaType(StateMachine, MessageBus, Storage, Time);
50
+ pub const Client = vsr.Client(StateMachine, MessageBus);
51
+ pub const StateChecker = StateCheckerType(Client, Replica);
52
+ pub const StorageChecker = StorageCheckerType(Replica);
53
+
54
+ pub const Options = struct {
55
+ cluster_id: u32,
56
+ replica_count: u8,
57
+ client_count: u8,
58
+ storage_size_limit: u64,
59
+ storage_fault_atlas: StorageFaultAtlas.Options,
60
+ seed: u64,
61
+
62
+ network: NetworkOptions,
63
+ storage: Storage.Options,
64
+ state_machine: StateMachine.Options,
65
+ };
66
+
67
+ allocator: mem.Allocator,
68
+ options: Options,
69
+ on_client_reply: fn (
70
+ cluster: *Self,
71
+ client: usize,
72
+ request: *Message,
73
+ reply: *Message,
74
+ ) void,
75
+
76
+ network: *Network,
77
+ storages: []Storage,
78
+ storage_fault_atlas: *StorageFaultAtlas,
79
+ replicas: []Replica,
80
+ replica_pools: []MessagePool,
81
+ replica_health: []ReplicaHealth,
82
+
83
+ clients: []Client,
84
+ client_pools: []MessagePool,
85
+ client_id_permutation: IdPermutation,
86
+
87
+ state_checker: StateChecker,
88
+ storage_checker: StorageChecker,
89
+
90
+ context: ?*anyopaque = null,
91
+
92
+ pub fn init(
93
+ allocator: mem.Allocator,
94
+ /// Includes command=register messages.
95
+ on_client_reply: fn (
96
+ cluster: *Self,
97
+ client: usize,
98
+ request: *Message,
99
+ reply: *Message,
100
+ ) void,
101
+ options: Options,
102
+ ) !*Self {
103
+ assert(options.replica_count >= 1);
104
+ assert(options.replica_count <= 6);
105
+ assert(options.client_count > 0);
106
+ assert(options.storage_size_limit % constants.sector_size == 0);
107
+ assert(options.storage_size_limit <= constants.storage_size_max);
108
+ assert(options.storage.replica_index == null);
109
+ assert(options.storage.fault_atlas == null);
110
+
111
+ var prng = std.rand.DefaultPrng.init(options.seed);
112
+ const random = prng.random();
113
+
114
+ // TODO(Zig) Client.init()'s MessagePool.Options require a reference to the network — use
115
+ // @returnAddress() instead.
116
+ var network = try allocator.create(Network);
117
+ errdefer allocator.destroy(network);
118
+
119
+ network.* = try Network.init(
120
+ allocator,
121
+ options.replica_count,
122
+ options.client_count,
123
+ options.network,
124
+ );
125
+ errdefer network.deinit();
126
+
127
+ // TODO(Zig) @returnAddress()
128
+ var storage_fault_atlas = try allocator.create(StorageFaultAtlas);
129
+ errdefer allocator.destroy(storage_fault_atlas);
130
+
131
+ storage_fault_atlas.* = StorageFaultAtlas.init(
132
+ options.replica_count,
133
+ random,
134
+ options.storage_fault_atlas,
135
+ );
136
+
137
+ const storages = try allocator.alloc(Storage, options.replica_count);
138
+ errdefer allocator.free(storages);
139
+
140
+ for (storages) |*storage, replica_index| {
141
+ var storage_options = options.storage;
142
+ storage_options.replica_index = @intCast(u8, replica_index);
143
+ storage_options.fault_atlas = storage_fault_atlas;
144
+ storage.* = try Storage.init(allocator, options.storage_size_limit, storage_options);
145
+ // Disable most faults at startup, so that the replicas don't get stuck recovering_head.
146
+ storage.faulty = replica_index >= vsr.quorums(options.replica_count).view_change;
147
+ }
148
+ errdefer for (storages) |*storage| storage.deinit(allocator);
149
+
150
+ var replica_pools = try allocator.alloc(MessagePool, options.replica_count);
151
+ errdefer allocator.free(replica_pools);
152
+
153
+ for (replica_pools) |*pool, i| {
154
+ errdefer for (replica_pools[0..i]) |*p| p.deinit(allocator);
155
+ pool.* = try MessagePool.init(allocator, .replica);
156
+ }
157
+ errdefer for (replica_pools) |*pool| pool.deinit(allocator);
158
+
159
+ const replicas = try allocator.alloc(Replica, options.replica_count);
160
+ errdefer allocator.free(replicas);
161
+
162
+ const replica_health = try allocator.alloc(ReplicaHealth, options.replica_count);
163
+ errdefer allocator.free(replica_health);
164
+ mem.set(ReplicaHealth, replica_health, .up);
165
+
166
+ var client_pools = try allocator.alloc(MessagePool, options.client_count);
167
+ errdefer allocator.free(client_pools);
168
+
169
+ for (client_pools) |*pool, i| {
170
+ errdefer for (client_pools[0..i]) |*p| p.deinit(allocator);
171
+ pool.* = try MessagePool.init(allocator, .client);
172
+ }
173
+ errdefer for (replica_pools) |*pool| pool.deinit(allocator);
174
+
175
+ const client_id_permutation = IdPermutation.generate(random);
176
+ var clients = try allocator.alloc(Client, options.client_count);
177
+ errdefer allocator.free(clients);
178
+
179
+ for (clients) |*client, i| {
180
+ errdefer for (clients[0..i]) |*c| c.deinit(allocator);
181
+ client.* = try Client.init(
182
+ allocator,
183
+ client_id_permutation.encode(i + client_id_permutation_shift),
184
+ options.cluster_id,
185
+ options.replica_count,
186
+ &client_pools[i],
187
+ .{ .network = network },
188
+ );
189
+ network.link(client.message_bus.process, &client.message_bus);
190
+ }
191
+ errdefer for (clients) |*c| c.deinit(allocator);
192
+
193
+ var state_checker =
194
+ try StateChecker.init(allocator, options.cluster_id, replicas, clients);
195
+ errdefer state_checker.deinit();
196
+
197
+ var storage_checker = StorageChecker.init(allocator);
198
+ errdefer storage_checker.deinit();
199
+
200
+ // Format each replica's storage (equivalent to "tigerbeetle format ...").
201
+ for (storages) |*storage, replica_index| {
202
+ var superblock = try SuperBlock.init(allocator, .{
203
+ .storage = storage,
204
+ .message_pool = &replica_pools[replica_index],
205
+ .storage_size_limit = options.storage_size_limit,
206
+ });
207
+ defer superblock.deinit(allocator);
208
+
209
+ try vsr.format(
210
+ Storage,
211
+ allocator,
212
+ options.cluster_id,
213
+ @intCast(u8, replica_index),
214
+ storage,
215
+ &superblock,
216
+ );
217
+ }
218
+
219
+ // We must heap-allocate the cluster since its pointer will be attached to the replica.
220
+ // TODO(Zig) @returnAddress().
221
+ var cluster = try allocator.create(Self);
222
+ errdefer allocator.destroy(cluster);
223
+
224
+ cluster.* = Self{
225
+ .allocator = allocator,
226
+ .options = options,
227
+ .on_client_reply = on_client_reply,
228
+ .network = network,
229
+ .storages = storages,
230
+ .storage_fault_atlas = storage_fault_atlas,
231
+ .replicas = replicas,
232
+ .replica_pools = replica_pools,
233
+ .replica_health = replica_health,
234
+ .clients = clients,
235
+ .client_pools = client_pools,
236
+ .client_id_permutation = client_id_permutation,
237
+ .state_checker = state_checker,
238
+ .storage_checker = storage_checker,
239
+ };
240
+
241
+ for (cluster.replicas) |_, replica_index| {
242
+ try cluster.open_replica(@intCast(u8, replica_index), .{
243
+ .resolution = constants.tick_ms * std.time.ns_per_ms,
244
+ .offset_type = .linear,
245
+ .offset_coefficient_A = 0,
246
+ .offset_coefficient_B = 0,
247
+ });
248
+ }
249
+ errdefer for (cluster.replicas) |*replica| replica.deinit(allocator);
250
+
251
+ return cluster;
252
+ }
253
+
254
+ pub fn deinit(cluster: *Self) void {
255
+ cluster.storage_checker.deinit();
256
+ cluster.state_checker.deinit();
257
+ cluster.network.deinit();
258
+ for (cluster.clients) |*client| client.deinit(cluster.allocator);
259
+ for (cluster.client_pools) |*pool| pool.deinit(cluster.allocator);
260
+ for (cluster.replicas) |*replica| replica.deinit(cluster.allocator);
261
+ for (cluster.replica_pools) |*pool| pool.deinit(cluster.allocator);
262
+ for (cluster.storages) |*storage| storage.deinit(cluster.allocator);
263
+
264
+ cluster.allocator.free(cluster.clients);
265
+ cluster.allocator.free(cluster.client_pools);
266
+ cluster.allocator.free(cluster.replicas);
267
+ cluster.allocator.free(cluster.replica_health);
268
+ cluster.allocator.free(cluster.replica_pools);
269
+ cluster.allocator.free(cluster.storages);
270
+ cluster.allocator.destroy(cluster.storage_fault_atlas);
271
+ cluster.allocator.destroy(cluster.network);
272
+ cluster.allocator.destroy(cluster);
273
+ }
274
+
275
+ pub fn tick(cluster: *Self) void {
276
+ cluster.network.tick();
277
+
278
+ for (cluster.clients) |*client| client.tick();
279
+ for (cluster.storages) |*storage| storage.tick();
280
+ for (cluster.replicas) |*replica, i| {
281
+ switch (cluster.replica_health[i]) {
282
+ .up => replica.tick(),
283
+ // Keep ticking the time so that it won't have diverged too far to synchronize
284
+ // when the replica restarts.
285
+ .down => replica.clock.time.tick(),
286
+ }
287
+ on_replica_change_state(replica);
288
+ }
289
+ }
290
+
291
+ pub fn restart_replica(cluster: *Self, replica_index: u8) void {
292
+ assert(cluster.replica_health[replica_index] == .down);
293
+
294
+ cluster.network.process_enable(.{ .replica = replica_index });
295
+ cluster.replica_health[replica_index] = .up;
296
+ }
297
+
298
+ /// Reset a replica to its initial state, simulating a random crash/panic.
299
+ /// Leave the persistent storage untouched, and leave any currently
300
+ /// inflight messages to/from the replica in the network.
301
+ ///
302
+ /// Returns whether the replica was crashed.
303
+ /// Returns an error when the replica was unable to recover (open).
304
+ pub fn crash_replica(cluster: *Self, replica_index: u8) !void {
305
+ assert(cluster.replica_health[replica_index] == .up);
306
+
307
+ // Reset the storage before the replica so that pending writes can (partially) finish.
308
+ cluster.storages[replica_index].reset();
309
+
310
+ const replica = &cluster.replicas[replica_index];
311
+ const replica_time = replica.time;
312
+ replica.deinit(cluster.allocator);
313
+ cluster.network.process_disable(.{ .replica = replica_index });
314
+ cluster.replica_health[replica_index] = .down;
315
+
316
+ // Ensure that none of the replica's messages leaked when it was deinitialized.
317
+ var messages_in_pool: usize = 0;
318
+ const message_bus = cluster.network.get_message_bus(.{ .replica = replica_index });
319
+ {
320
+ var it = message_bus.pool.free_list;
321
+ while (it) |message| : (it = message.next) messages_in_pool += 1;
322
+ }
323
+ assert(messages_in_pool == message_pool.messages_max_replica);
324
+
325
+ // Logically it would make more sense to run this during restart, not immediately following
326
+ // the crash. But having it here allows the replica's MessageBus to initialize and begin
327
+ // queueing packets.
328
+ //
329
+ // Pass the old replica's Time through to the new replica. It will continue to tick while
330
+ // the replica is crashed, to ensure the clocks don't desyncronize too far to recover.
331
+ try cluster.open_replica(replica_index, replica_time);
332
+ }
333
+
334
+ fn open_replica(cluster: *Self, replica_index: u8, time: Time) !void {
335
+ var replica = &cluster.replicas[replica_index];
336
+ try replica.open(
337
+ cluster.allocator,
338
+ .{
339
+ .replica_count = @intCast(u8, cluster.replicas.len),
340
+ .storage = &cluster.storages[replica_index],
341
+ // TODO Test restarting with a higher storage limit.
342
+ .storage_size_limit = cluster.options.storage_size_limit,
343
+ .message_pool = &cluster.replica_pools[replica_index],
344
+ .time = time,
345
+ .state_machine_options = cluster.options.state_machine,
346
+ .message_bus_options = .{ .network = cluster.network },
347
+ },
348
+ );
349
+ assert(replica.cluster == cluster.options.cluster_id);
350
+ assert(replica.replica == replica_index);
351
+ assert(replica.replica_count == cluster.replicas.len);
352
+
353
+ replica.context = cluster;
354
+ replica.on_change_state = on_replica_change_state;
355
+ replica.on_compact = on_replica_compact;
356
+ replica.on_checkpoint = on_replica_checkpoint;
357
+ cluster.network.link(replica.message_bus.process, &replica.message_bus);
358
+ }
359
+
360
+ pub fn request(
361
+ cluster: *Self,
362
+ client_index: usize,
363
+ request_operation: StateMachine.Operation,
364
+ request_message: *Message,
365
+ request_body_size: usize,
366
+ ) void {
367
+ // TODO(Zig) Move these into init when `@returnAddress()` is available. They only needs to
368
+ // be set once, it just requires a stable pointer to the Cluster.
369
+ cluster.clients[client_index].on_reply_context = cluster;
370
+ cluster.clients[client_index].on_reply_callback = client_on_reply;
371
+
372
+ cluster.clients[client_index].request(
373
+ undefined,
374
+ request_callback,
375
+ request_operation,
376
+ request_message,
377
+ request_body_size,
378
+ );
379
+ }
380
+
381
+ /// The `request_callback` is not used — Cluster uses `Client.on_reply_{context,callback}`
382
+ /// instead because:
383
+ /// - Cluster needs access to the request
384
+ /// - Cluster needs access to the reply message (not just the body)
385
+ /// - Cluster needs to know about command=register messages
386
+ ///
387
+ /// See `on_reply`.
388
+ fn request_callback(
389
+ user_data: u128,
390
+ operation: StateMachine.Operation,
391
+ result: Client.Error![]const u8,
392
+ ) void {
393
+ _ = user_data;
394
+ _ = operation;
395
+ _ = result catch |err| switch (err) {
396
+ error.TooManyOutstandingRequests => unreachable,
397
+ };
398
+ }
399
+
400
+ fn client_on_reply(client: *Client, request_message: *Message, reply_message: *Message) void {
401
+ const cluster = @ptrCast(*Self, @alignCast(@alignOf(Self), client.on_reply_context.?));
402
+ assert(reply_message.header.cluster == cluster.options.cluster_id);
403
+ assert(reply_message.header.invalid() == null);
404
+ assert(reply_message.header.client == client.id);
405
+ assert(reply_message.header.request == request_message.header.request);
406
+ assert(reply_message.header.command == .reply);
407
+ assert(reply_message.header.operation == request_message.header.operation);
408
+
409
+ const client_index = for (cluster.clients) |*c, i| {
410
+ if (client == c) break i;
411
+ } else unreachable;
412
+
413
+ cluster.on_client_reply(cluster, client_index, request_message, reply_message);
414
+ }
415
+
416
+ fn on_replica_change_state(replica: *const Replica) void {
417
+ const cluster = @ptrCast(*Self, @alignCast(@alignOf(Self), replica.context.?));
418
+ cluster.state_checker.check_state(replica.replica) catch |err| {
419
+ fatal(.correctness, "state checker error: {}", .{err});
420
+ };
421
+ }
422
+
423
+ fn on_replica_compact(replica: *const Replica) void {
424
+ const cluster = @ptrCast(*Self, @alignCast(@alignOf(Self), replica.context.?));
425
+ cluster.storage_checker.replica_compact(replica) catch |err| {
426
+ fatal(.correctness, "storage checker error: {}", .{err});
427
+ };
428
+ }
429
+
430
+ fn on_replica_checkpoint(replica: *const Replica) void {
431
+ const cluster = @ptrCast(*Self, @alignCast(@alignOf(Self), replica.context.?));
432
+ cluster.storage_checker.replica_checkpoint(replica) catch |err| {
433
+ fatal(.correctness, "storage checker error: {}", .{err});
434
+ };
435
+ }
436
+
437
+ /// Print an error message and then exit with an exit code.
438
+ fn fatal(failure: Failure, comptime fmt_string: []const u8, args: anytype) noreturn {
439
+ std.log.scoped(.state_checker).err(fmt_string, args);
440
+ std.os.exit(@enumToInt(failure));
441
+ }
442
+ };
443
+ }
File without changes
@@ -0,0 +1,66 @@
1
+ //! A tool for narrowing down the point of divergence between two executions that should be identical.
2
+ //! Sprinkle calls to `emit(some_hash)` throughout the code.
3
+ //! With `-Dhash-log-mode=create`, all emitted hashes are written to ./hash_log.
4
+ //! With `-Dhash-log-mode=check`, all emitted hashes are checked against the hashes in ./hash_log.
5
+ //! Otherwise, calls to `emit` are noops.
6
+
7
+ const std = @import("std");
8
+ const assert = std.debug.assert;
9
+ const panic = std.debug.panic;
10
+
11
+ const constants = @import("../constants.zig");
12
+
13
+ var file: ?std.fs.File = null;
14
+ var hash_count: usize = 0;
15
+
16
+ fn ensure_init() void {
17
+ if (file != null) return;
18
+ switch (constants.hash_log_mode) {
19
+ .none => unreachable,
20
+ .create => {
21
+ file = std.fs.cwd().createFile("./hash_log", .{ .truncate = true }) catch unreachable;
22
+ },
23
+ .check => {
24
+ file = std.fs.cwd().openFile("./hash_log", .{ .read = true }) catch unreachable;
25
+ },
26
+ }
27
+ }
28
+
29
+ pub fn emit(hash: u128) void {
30
+ @call(.{ .modifier = .never_inline }, emit_never_inline, .{hash});
31
+ }
32
+
33
+ // Don't inline because we want to be able to break on this function.
34
+ fn emit_never_inline(hash: u128) void {
35
+ switch (constants.hash_log_mode) {
36
+ .none => {},
37
+ .create => {
38
+ ensure_init();
39
+ std.fmt.format(file.?.writer(), "{x:0>32}\n", .{hash}) catch unreachable;
40
+ hash_count += 1;
41
+ },
42
+ .check => {
43
+ ensure_init();
44
+ var buffer: [33]u8 = undefined;
45
+ const bytes_read = file.?.readAll(&buffer) catch unreachable;
46
+ if (bytes_read != 33) {
47
+ panic("Unexpected end of hash_log at hash_count={}. Expected EOF, found {x:0>32}.", .{ hash_count, hash });
48
+ }
49
+ const expected_hash = std.fmt.parseInt(u128, buffer[0..32], 16) catch unreachable;
50
+ if (hash != expected_hash) {
51
+ panic(
52
+ "Hash mismatch at hash_count={}. Expected {x:0>32}, found {x:0>32}.",
53
+ .{ hash_count, expected_hash, hash },
54
+ );
55
+ }
56
+ hash_count += 1;
57
+ },
58
+ }
59
+ }
60
+
61
+ pub fn emit_autohash(hashable: anytype, comptime strategy: std.hash.Strategy) void {
62
+ if (constants.hash_log_mode == .none) return;
63
+ var hasher = std.hash.Wyhash.init(0);
64
+ std.hash.autoHashStrat(&hasher, hashable, strategy);
65
+ emit(hasher.final());
66
+ }
File without changes