odba 1.2.1 → 1.2.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/History.md +48 -0
- data/lib/odba/cache.rb +14 -2
- data/lib/odba/storage.rb +63 -5
- data/lib/odba/version.rb +1 -1
- data/test/test_cache.rb +70 -0
- data/test/test_storage.rb +29 -0
- metadata +1 -1
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: a5870eb15332e2b3e862dd27adf014e175e46588d673527794988a3a876f61b7
|
|
4
|
+
data.tar.gz: 8209e0f9122128f21e004f9db3a2fc7b58686f1c2b6f802ce3be628d095cd343
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: e7a633a8aeecdd7d243bce697800a3163b9090cae5352609aac8139704594bb02e71521f443829efeb618e1882c7f94b19b014c03b4e606c62ba66832566c173
|
|
7
|
+
data.tar.gz: abb1a74dff537de863797b397185e05ef1c6301da2d0e51ffcd48b3621b6a89f5933534b4fca02067d305abcb5645988b8eb805a07432d4e9e358fbf9d2cf7ea
|
data/History.md
CHANGED
|
@@ -1,3 +1,51 @@
|
|
|
1
|
+
## 1.2.3 / 01.09.2026
|
|
2
|
+
|
|
3
|
+
1.2.2 narrowed the rescue around the peer notification in `Cache#next_id` too
|
|
4
|
+
far. "A peer we cannot reach" is not only `DRb::DRbError`: a reference into a
|
|
5
|
+
peer that has restarted, or whose entry has expired from the DRb object space,
|
|
6
|
+
raises `RangeError("invalid reference")` from `DRbObjectSpace#to_obj` - a
|
|
7
|
+
StandardError, and not a DRbError. 1.1.9's classless rescue absorbed it; 1.2.2
|
|
8
|
+
let it out and it reached the caller.
|
|
9
|
+
|
|
10
|
+
In oddb.org that showed up on the first night under 1.2.2: three index
|
|
11
|
+
rebuilds died with `invalid reference (druby://127.0.0.1:10000)`. An index
|
|
12
|
+
that is not built is a deferred index, and ODBA fills a deferred index in
|
|
13
|
+
`Cache#setup` - so the *next* process to start died on it too, before doing
|
|
14
|
+
any work of its own.
|
|
15
|
+
|
|
16
|
+
* `next_id` re-raises `OdbaDuplicateIdError` (the one exception that must
|
|
17
|
+
reach the retry, which is what 1.2.2 was for) and swallows every other
|
|
18
|
+
StandardError from a peer, as 1.1.9 did.
|
|
19
|
+
* Two regression tests, one per direction: a stale reference must not stop
|
|
20
|
+
the allocation, and the catch-all must not swallow the conflict again.
|
|
21
|
+
The first fails against 1.2.2 with the production error.
|
|
22
|
+
|
|
23
|
+
## 1.2.2 / 31.08.2026
|
|
24
|
+
|
|
25
|
+
Two processes on one database handed out the same odba_id. Whichever wrote
|
|
26
|
+
last overwrote the other's row in `object`, and every reference to the lost
|
|
27
|
+
object then resolved to a foreign one - an Array where a domain object
|
|
28
|
+
belonged, or the reverse. The referring instance variable stays correct and
|
|
29
|
+
points at the right number; a different object simply sits under it, so
|
|
30
|
+
searching the application for the offending assignment finds nothing.
|
|
31
|
+
|
|
32
|
+
* `Storage#next_id` takes the id from a Postgres sequence, `odba_id_seq`,
|
|
33
|
+
which `#setup` creates. It used to be `@next_id += 1` under a mutex, with
|
|
34
|
+
@next_id seeded once per process from the highest odba_id in the table -
|
|
35
|
+
sound for one process, wrong for every deployment running a web worker and
|
|
36
|
+
an import job against the same database. Measured with two processes side
|
|
37
|
+
by side, both answered `[61935067, 61935068, 61935069]`. Stores without the
|
|
38
|
+
sequence keep the old behaviour.
|
|
39
|
+
* The sequence starts at `MAX(odba_id)` plus `ID_SEQUENCE_GAP`, not at 1: a
|
|
40
|
+
plain `CREATE SEQUENCE` would re-issue ids that already exist, and the gap
|
|
41
|
+
covers ids that processes still on the old counter hold but have not
|
|
42
|
+
written yet.
|
|
43
|
+
* `Cache#next_id` no longer swallows `OdbaDuplicateIdError`. The guard was
|
|
44
|
+
there all along - a peer raises it when the id is taken and the method
|
|
45
|
+
retries - but `rescue` without a class caught it too, so the retry could
|
|
46
|
+
never run. Only `DRb::DRbError` is caught now, which is what the line was
|
|
47
|
+
for: an unreachable peer must not stop the allocation.
|
|
48
|
+
|
|
1
49
|
## 1.2.1 / 21.08.2026
|
|
2
50
|
|
|
3
51
|
No library changes: lib/ is identical to 1.2.0. This release only ships a test
|
data/lib/odba/cache.rb
CHANGED
|
@@ -441,8 +441,20 @@ module ODBA
|
|
|
441
441
|
end
|
|
442
442
|
@peers.each do |peer|
|
|
443
443
|
peer.reserve_next_id id
|
|
444
|
-
rescue
|
|
445
|
-
|
|
444
|
+
rescue OdbaDuplicateIdError
|
|
445
|
+
# The one exception that must get through, to the retry below. Until
|
|
446
|
+
# 1.2.2 a bare rescue swallowed it, so the retry could never run and
|
|
447
|
+
# both processes kept the same id; whichever wrote last overwrote the
|
|
448
|
+
# other's row in `object`, and every reference to it then resolved to
|
|
449
|
+
# a foreign object.
|
|
450
|
+
raise
|
|
451
|
+
rescue StandardError
|
|
452
|
+
# Anything else means we could not reach this peer, and that must not
|
|
453
|
+
# stop the allocation. Naming DRb::DRbError alone is not enough, which
|
|
454
|
+
# 1.2.2 got wrong: a stale reference into a peer raises RangeError
|
|
455
|
+
# ("invalid reference", drb.rb DRbObjectSpace#to_obj), which is not a
|
|
456
|
+
# DRbError. On 01.09.2026 that took down three index rebuilds in
|
|
457
|
+
# oddb.org - the peer had simply gone away.
|
|
446
458
|
end
|
|
447
459
|
id
|
|
448
460
|
rescue OdbaDuplicateIdError
|
data/lib/odba/storage.rb
CHANGED
|
@@ -9,8 +9,17 @@ require "dbi"
|
|
|
9
9
|
module ODBA
|
|
10
10
|
class Storage # :nodoc: all
|
|
11
11
|
include Singleton
|
|
12
|
-
|
|
12
|
+
|
|
13
|
+
# Not attr_writer: whether the store has an id sequence is memoized, and
|
|
14
|
+
# that answer belongs to the connection it was asked on.
|
|
15
|
+
def dbi=(dbi)
|
|
16
|
+
@id_sequence = nil
|
|
17
|
+
@dbi = dbi
|
|
18
|
+
end
|
|
13
19
|
BULK_FETCH_STEP = 2500
|
|
20
|
+
# Distance between the highest id in use and the first the sequence
|
|
21
|
+
# hands out; see the odba_id_seq entry in TABLES.
|
|
22
|
+
ID_SEQUENCE_GAP = 100_000
|
|
14
23
|
TABLES = [
|
|
15
24
|
# in table 'object', the isolated dumps of all objects are stored
|
|
16
25
|
["object", <<~SQL],
|
|
@@ -37,12 +46,31 @@ module ODBA
|
|
|
37
46
|
CREATE INDEX IF NOT EXISTS target_id_index ON object_connection(target_id);
|
|
38
47
|
SQL
|
|
39
48
|
# helper table 'collection'
|
|
40
|
-
["collection", <<~SQL]
|
|
49
|
+
["collection", <<~SQL],
|
|
41
50
|
CREATE TABLE IF NOT EXISTS collection (
|
|
42
51
|
odba_id integer NOT NULL, key text, value text,
|
|
43
52
|
PRIMARY KEY(odba_id, key)
|
|
44
53
|
);
|
|
45
54
|
SQL
|
|
55
|
+
# The odba_id comes from this sequence, so that several processes on
|
|
56
|
+
# one database cannot hand out the same one. See #next_id.
|
|
57
|
+
#
|
|
58
|
+
# The start value is computed and must be: a plain CREATE SEQUENCE
|
|
59
|
+
# starts at 1 and would re-issue ids that already exist. The gap on
|
|
60
|
+
# top of MAX(odba_id) covers ids that processes still running with the
|
|
61
|
+
# old in-memory counter hold but have not written yet. Skipped numbers
|
|
62
|
+
# cost nothing - the odba_id is a surrogate key and carries no meaning.
|
|
63
|
+
["odba_id_seq", <<~SQL]
|
|
64
|
+
DO $$
|
|
65
|
+
BEGIN
|
|
66
|
+
IF NOT EXISTS (SELECT 1 FROM pg_class
|
|
67
|
+
WHERE relkind = 'S' AND relname = 'odba_id_seq') THEN
|
|
68
|
+
EXECUTE format('CREATE SEQUENCE odba_id_seq START WITH %s',
|
|
69
|
+
(SELECT COALESCE(MAX(odba_id), 0) + #{ID_SEQUENCE_GAP}
|
|
70
|
+
FROM object));
|
|
71
|
+
END IF;
|
|
72
|
+
END $$;
|
|
73
|
+
SQL
|
|
46
74
|
]
|
|
47
75
|
def initialize
|
|
48
76
|
@id_mutex = Mutex.new
|
|
@@ -425,13 +453,43 @@ module ODBA
|
|
|
425
453
|
end
|
|
426
454
|
end
|
|
427
455
|
|
|
456
|
+
# The id is allocated by the database, not by a counter in this process.
|
|
457
|
+
#
|
|
458
|
+
# It used to be `@next_id += 1` under this mutex, with @next_id seeded
|
|
459
|
+
# once per process from the highest odba_id in the table. That is sound
|
|
460
|
+
# for a single process and wrong for every deployment that runs more
|
|
461
|
+
# than one - web workers and import jobs on the same database each kept
|
|
462
|
+
# their own counter and handed out the same numbers, so one silently
|
|
463
|
+
# overwrote the other's row in `object`.
|
|
464
|
+
#
|
|
465
|
+
# Falls back to the old behaviour where no sequence exists, so a store
|
|
466
|
+
# that was never through #setup keeps working; #setup creates it.
|
|
428
467
|
def next_id
|
|
429
|
-
|
|
430
|
-
|
|
431
|
-
|
|
468
|
+
if id_sequence?
|
|
469
|
+
dbi.select_one("SELECT nextval('odba_id_seq')").first.to_i.tap { |id|
|
|
470
|
+
# max_id and reserve_next_id read @next_id, so keep it in step.
|
|
471
|
+
# Never backwards: a peer may already stand higher.
|
|
472
|
+
@id_mutex.synchronize {
|
|
473
|
+
@next_id = id if @next_id.nil? || @next_id < id
|
|
474
|
+
}
|
|
475
|
+
}
|
|
476
|
+
else
|
|
477
|
+
@id_mutex.synchronize do
|
|
478
|
+
ensure_next_id_set
|
|
479
|
+
@next_id += 1
|
|
480
|
+
end
|
|
432
481
|
end
|
|
433
482
|
end
|
|
434
483
|
|
|
484
|
+
def id_sequence?
|
|
485
|
+
return @id_sequence unless @id_sequence.nil?
|
|
486
|
+
@id_sequence = dbi.select_one(
|
|
487
|
+
"SELECT 1 FROM pg_class WHERE relkind = 'S' AND relname = 'odba_id_seq'"
|
|
488
|
+
) ? true : false
|
|
489
|
+
rescue
|
|
490
|
+
@id_sequence = false
|
|
491
|
+
end
|
|
492
|
+
|
|
435
493
|
def update_max_id(id)
|
|
436
494
|
@id_mutex.synchronize do
|
|
437
495
|
@next_id = id
|
data/lib/odba/version.rb
CHANGED
data/test/test_cache.rb
CHANGED
|
@@ -19,6 +19,76 @@ module ODBA
|
|
|
19
19
|
public :load_object
|
|
20
20
|
end
|
|
21
21
|
|
|
22
|
+
# The peer conflict must reach the retry. Until 1.2.2 the loop read
|
|
23
|
+
# `peer.reserve_next_id id rescue DRb::DRbError` - a rescue without a
|
|
24
|
+
# class, which caught the OdbaDuplicateIdError a peer raises when the id
|
|
25
|
+
# is taken. Both processes then kept the same id and one overwrote the
|
|
26
|
+
# other's row in `object`.
|
|
27
|
+
class TestCacheNextId < Test::Unit::TestCase
|
|
28
|
+
include FlexMock::TestCase
|
|
29
|
+
|
|
30
|
+
def setup
|
|
31
|
+
@cache = ODBA::Cache.instance
|
|
32
|
+
@cache.instance_variable_set(:@file_lock, false)
|
|
33
|
+
@storage = flexmock("storage")
|
|
34
|
+
ODBA.storage = @storage
|
|
35
|
+
end
|
|
36
|
+
|
|
37
|
+
def test_a_peer_conflict_leads_to_a_new_id
|
|
38
|
+
@storage.should_receive(:next_id).and_return(100, 101)
|
|
39
|
+
seen = []
|
|
40
|
+
peer = flexmock("peer")
|
|
41
|
+
peer.should_receive(:reserve_next_id).and_return { |id|
|
|
42
|
+
seen << id
|
|
43
|
+
raise ODBA::OdbaDuplicateIdError, "taken" if seen.size == 1
|
|
44
|
+
true
|
|
45
|
+
}
|
|
46
|
+
@cache.instance_variable_set(:@peers, [peer])
|
|
47
|
+
assert_equal(101, @cache.next_id)
|
|
48
|
+
assert_equal([100, 101], seen)
|
|
49
|
+
end
|
|
50
|
+
|
|
51
|
+
# A peer we cannot reach must not stop the allocation - that is what the
|
|
52
|
+
# line was for, and it stays.
|
|
53
|
+
def test_an_unreachable_peer_does_not_stop_the_allocation
|
|
54
|
+
@storage.should_receive(:next_id).and_return(100)
|
|
55
|
+
peer = flexmock("peer")
|
|
56
|
+
peer.should_receive(:reserve_next_id).and_raise(DRb::DRbError, "gone")
|
|
57
|
+
@cache.instance_variable_set(:@peers, [peer])
|
|
58
|
+
assert_equal(100, @cache.next_id)
|
|
59
|
+
end
|
|
60
|
+
|
|
61
|
+
# And "cannot reach" is not only DRbError. A reference into a peer that
|
|
62
|
+
# has restarted or expired raises RangeError("invalid reference") from
|
|
63
|
+
# DRbObjectSpace#to_obj - a StandardError, not a DRbError. 1.2.2 named
|
|
64
|
+
# DRbError alone and so let it out: on 01.09.2026 three index rebuilds in
|
|
65
|
+
# oddb.org died on it, which 1.1.9's classless rescue had absorbed.
|
|
66
|
+
def test_a_stale_reference_into_a_peer_does_not_stop_the_allocation
|
|
67
|
+
@storage.should_receive(:next_id).and_return(100)
|
|
68
|
+
peer = flexmock("peer")
|
|
69
|
+
peer.should_receive(:reserve_next_id)
|
|
70
|
+
.and_raise(RangeError, "invalid reference")
|
|
71
|
+
@cache.instance_variable_set(:@peers, [peer])
|
|
72
|
+
assert_equal(100, @cache.next_id)
|
|
73
|
+
end
|
|
74
|
+
|
|
75
|
+
# The catch-all must not grow so wide that it takes the conflict with it
|
|
76
|
+
# again - the whole point of 1.2.2. One test for each direction.
|
|
77
|
+
def test_a_conflict_is_not_swallowed_by_the_catch_all
|
|
78
|
+
@storage.should_receive(:next_id).and_return(100, 101)
|
|
79
|
+
calls = 0
|
|
80
|
+
peer = flexmock("peer")
|
|
81
|
+
peer.should_receive(:reserve_next_id).and_return {
|
|
82
|
+
calls += 1
|
|
83
|
+
raise ODBA::OdbaDuplicateIdError, "taken" if calls == 1
|
|
84
|
+
true
|
|
85
|
+
}
|
|
86
|
+
@cache.instance_variable_set(:@peers, [peer])
|
|
87
|
+
assert_equal(101, @cache.next_id)
|
|
88
|
+
assert_equal(2, calls)
|
|
89
|
+
end
|
|
90
|
+
end
|
|
91
|
+
|
|
22
92
|
class TestCache < Test::Unit::TestCase
|
|
23
93
|
include FlexMock::TestCase
|
|
24
94
|
class ODBAContainerInCache
|
data/test/test_storage.rb
CHANGED
|
@@ -15,6 +15,7 @@ module ODBA
|
|
|
15
15
|
@storage = ODBA::Storage.instance
|
|
16
16
|
@dbi = flexmock("DBI")
|
|
17
17
|
@storage.dbi = @dbi
|
|
18
|
+
@storage.instance_variable_set(:@id_sequence, nil)
|
|
18
19
|
end
|
|
19
20
|
|
|
20
21
|
def teardown
|
|
@@ -106,12 +107,40 @@ module ODBA
|
|
|
106
107
|
@storage.create_index("index_name")
|
|
107
108
|
end
|
|
108
109
|
|
|
110
|
+
# Without a sequence the old counter still runs, so a store that was
|
|
111
|
+
# never through #setup keeps working.
|
|
109
112
|
def test_next_id
|
|
113
|
+
@dbi.should_receive(:select_one).and_return(nil)
|
|
110
114
|
@storage.next_id = 1
|
|
111
115
|
assert_equal(2, @storage.next_id)
|
|
112
116
|
assert_equal(3, @storage.next_id)
|
|
113
117
|
end
|
|
114
118
|
|
|
119
|
+
# The point of 1.2.2: the id comes from the database, so two processes
|
|
120
|
+
# on one store cannot be handed the same one.
|
|
121
|
+
def test_next_id__from_the_sequence
|
|
122
|
+
@dbi.should_receive(:select_one)
|
|
123
|
+
.with("SELECT 1 FROM pg_class WHERE relkind = 'S' AND relname = 'odba_id_seq'")
|
|
124
|
+
.once.and_return([1])
|
|
125
|
+
@dbi.should_receive(:select_one)
|
|
126
|
+
.with("SELECT nextval('odba_id_seq')").twice.and_return([4711], [4712])
|
|
127
|
+
assert_equal(4711, @storage.next_id)
|
|
128
|
+
assert_equal(4712, @storage.next_id)
|
|
129
|
+
end
|
|
130
|
+
|
|
131
|
+
# max_id and reserve_next_id read @next_id, so it has to follow the
|
|
132
|
+
# sequence - and never move backwards, a peer may stand higher already.
|
|
133
|
+
def test_next_id__keeps_the_local_counter_in_step
|
|
134
|
+
@dbi.should_receive(:select_one)
|
|
135
|
+
.with(/pg_class/).and_return([1])
|
|
136
|
+
@dbi.should_receive(:select_one).with(/nextval/).and_return([90], [5])
|
|
137
|
+
@storage.next_id = 10
|
|
138
|
+
@storage.next_id
|
|
139
|
+
assert_equal(90, @storage.instance_variable_get(:@next_id))
|
|
140
|
+
@storage.next_id
|
|
141
|
+
assert_equal(90, @storage.instance_variable_get(:@next_id))
|
|
142
|
+
end
|
|
143
|
+
|
|
115
144
|
def test_store__1
|
|
116
145
|
dbi = flexmock("dbi")
|
|
117
146
|
@storage.dbi = dbi
|