rubydb 0.1.5 → 0.1.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.gitignore +4 -0
- data/CHANGELOG.md +14 -0
- data/Gemfile.lock +1 -1
- data/README.md +296 -227
- data/Rakefile +6 -1
- data/accelerator/bin/SHA256SUMS +6 -0
- data/accelerator/bin/rubydb-accelerator-darwin-amd64 +0 -0
- data/accelerator/bin/rubydb-accelerator-darwin-arm64 +0 -0
- data/accelerator/bin/rubydb-accelerator-linux-amd64 +0 -0
- data/accelerator/bin/rubydb-accelerator-linux-arm64 +0 -0
- data/accelerator/bin/rubydb-accelerator-windows-amd64.exe +0 -0
- data/accelerator/bin/rubydb-accelerator-windows-arm64.exe +0 -0
- data/accelerator/cmd/rubydb-accelerator/main.go +11 -0
- data/accelerator/go.mod +3 -0
- data/accelerator/internal/execution/aggregate.go +94 -0
- data/accelerator/internal/execution/distinct.go +22 -0
- data/accelerator/internal/execution/filter.go +73 -0
- data/accelerator/internal/execution/join.go +79 -0
- data/accelerator/internal/execution/operators.go +167 -0
- data/accelerator/internal/execution/scan.go +20 -0
- data/accelerator/internal/execution/sort.go +62 -0
- data/accelerator/internal/execution/types.go +136 -0
- data/accelerator/internal/execution/value.go +67 -0
- data/accelerator/internal/memory/arena.go +47 -0
- data/accelerator/internal/memory/reuse.go +22 -0
- data/accelerator/internal/metrics/registry.go +67 -0
- data/accelerator/internal/parallel/bounded_queue.go +56 -0
- data/accelerator/internal/parallel/scheduler.go +47 -0
- data/accelerator/internal/parallel/worker_pool.go +53 -0
- data/accelerator/internal/protocol/cancellation.go +48 -0
- data/accelerator/internal/protocol/columnar.go +263 -0
- data/accelerator/internal/protocol/frame.go +187 -0
- data/accelerator/internal/runtime/worker.go +521 -0
- data/accelerator/internal/storage/page_reader.go +81 -0
- data/accelerator/internal/storage/snapshot_scan.go +539 -0
- data/accelerator/internal/wal/checksum.go +13 -0
- data/accelerator/internal/wal/compression.go +41 -0
- data/accelerator/internal/wal/group_commit.go +24 -0
- data/accelerator/internal/wal/record_encoder.go +40 -0
- data/adapters/activerecord/README.md +8 -3
- data/adapters/activerecord/lib/active_record/connection_adapters/rubydb_adapter.rb +50 -42
- data/adapters/activerecord/rubydb-activerecord.gemspec +1 -1
- data/docs/README.md +3 -1
- data/docs/architecture/go-accelerator.md +179 -0
- data/docs/cli.md +24 -0
- data/docs/contributing/benchmarking.md +16 -0
- data/docs/developer/local-development.md +32 -0
- data/docs/release.md +2 -2
- data/lessons/02-local-development.md +2 -2
- data/lessons/04-rails-complex-apps.md +2 -2
- data/lessons/05-rubydb-production-server.md +2 -2
- data/lessons/07-hybrid-microservices.md +175 -90
- data/lessons/10-release-readiness.md +184 -117
- data/lessons/11-community-adapter.md +323 -0
- data/lessons/12-rails-ecommerce-pressure.md +263 -0
- data/lib/rubydb/accelerator/client.rb +451 -0
- data/lib/rubydb/accelerator/error.rb +22 -0
- data/lib/rubydb/accelerator/manager.rb +606 -0
- data/lib/rubydb/accelerator.rb +13 -0
- data/lib/rubydb/cli/application.rb +6 -1
- data/lib/rubydb/cli/commands/accelerator.rb +72 -0
- data/lib/rubydb/cli/commands/doctor.rb +3 -0
- data/lib/rubydb/client/client.rb +7 -0
- data/lib/rubydb/client/connection.rb +15 -0
- data/lib/rubydb/client/result.rb +5 -1
- data/lib/rubydb/configuration/defaults.rb +12 -0
- data/lib/rubydb/configuration/validation.rb +8 -1
- data/lib/rubydb/execution/accelerator_dispatch.rb +30 -0
- data/lib/rubydb/execution/cost_model.rb +72 -0
- data/lib/rubydb/execution/executor.rb +373 -11
- data/lib/rubydb/execution/operator_selection.rb +57 -0
- data/lib/rubydb/execution/physical_plan.rb +47 -0
- data/lib/rubydb/execution/planner.rb +12 -46
- data/lib/rubydb/execution/sort_executor.rb +22 -8
- data/lib/rubydb/indexes/btree.rb +31 -2
- data/lib/rubydb/rubydb.rb +7 -1
- data/lib/rubydb/server/session.rb +45 -0
- data/lib/rubydb/storage/engine.rb +74 -12
- data/lib/rubydb/storage/snapshot_reader.rb +167 -0
- data/lib/rubydb/version.rb +1 -1
- data/lib/rubydb/wal/archive.rb +17 -0
- data/lib/rubydb/wal/wal.rb +1 -0
- data/rubydb.gemspec +12 -2
- data/scripts/build_accelerator +49 -0
- data/scripts/release +34 -4
- data/scripts/replication_failover_drill +2 -2
- metadata +49 -1
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
package storage
|
|
2
|
+
|
|
3
|
+
import (
|
|
4
|
+
"context"
|
|
5
|
+
"crypto/sha256"
|
|
6
|
+
"errors"
|
|
7
|
+
"io"
|
|
8
|
+
"sync"
|
|
9
|
+
)
|
|
10
|
+
|
|
11
|
+
var ErrInvalidPage = errors.New("invalid storage page")
|
|
12
|
+
|
|
13
|
+
type Snapshot struct {
|
|
14
|
+
ID uint64
|
|
15
|
+
PageSize int
|
|
16
|
+
MaxPages uint64
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
type Page struct {
|
|
20
|
+
Number uint64
|
|
21
|
+
Data []byte
|
|
22
|
+
Digest [32]byte
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
type Reader struct {
|
|
26
|
+
input io.ReaderAt
|
|
27
|
+
cache sync.Map
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
func NewReader(input io.ReaderAt) *Reader { return &Reader{input: input} }
|
|
31
|
+
|
|
32
|
+
// ReadBatch reads immutable pages for a snapshot. The caller must provide a
|
|
33
|
+
// snapshot validated by RubyDB; this package never reads catalog or WAL state.
|
|
34
|
+
func (reader *Reader) ReadBatch(ctx context.Context, snapshot Snapshot, pageNumbers []uint64) ([]Page, error) {
|
|
35
|
+
if snapshot.PageSize <= 0 || snapshot.MaxPages == 0 || uint64(snapshot.PageSize) > 64*1024*1024 {
|
|
36
|
+
return nil, ErrInvalidPage
|
|
37
|
+
}
|
|
38
|
+
result := make([]Page, len(pageNumbers))
|
|
39
|
+
var wait sync.WaitGroup
|
|
40
|
+
var firstErr error
|
|
41
|
+
var errMu sync.Mutex
|
|
42
|
+
for index, number := range pageNumbers {
|
|
43
|
+
if number >= snapshot.MaxPages {
|
|
44
|
+
return nil, ErrInvalidPage
|
|
45
|
+
}
|
|
46
|
+
wait.Add(1)
|
|
47
|
+
go func(index int, number uint64) {
|
|
48
|
+
defer wait.Done()
|
|
49
|
+
select {
|
|
50
|
+
case <-ctx.Done():
|
|
51
|
+
errMu.Lock()
|
|
52
|
+
if firstErr == nil {
|
|
53
|
+
firstErr = ctx.Err()
|
|
54
|
+
}
|
|
55
|
+
errMu.Unlock()
|
|
56
|
+
return
|
|
57
|
+
default:
|
|
58
|
+
}
|
|
59
|
+
data := make([]byte, snapshot.PageSize)
|
|
60
|
+
read, err := reader.input.ReadAt(data, int64(number)*int64(snapshot.PageSize))
|
|
61
|
+
if read != snapshot.PageSize || (err != nil && err != io.EOF) {
|
|
62
|
+
errMu.Lock()
|
|
63
|
+
if firstErr == nil {
|
|
64
|
+
if err != nil {
|
|
65
|
+
firstErr = err
|
|
66
|
+
} else {
|
|
67
|
+
firstErr = ErrInvalidPage
|
|
68
|
+
}
|
|
69
|
+
}
|
|
70
|
+
errMu.Unlock()
|
|
71
|
+
return
|
|
72
|
+
}
|
|
73
|
+
result[index] = Page{Number: number, Data: data, Digest: sha256.Sum256(data)}
|
|
74
|
+
}(index, number)
|
|
75
|
+
}
|
|
76
|
+
wait.Wait()
|
|
77
|
+
if firstErr != nil {
|
|
78
|
+
return nil, firstErr
|
|
79
|
+
}
|
|
80
|
+
return result, nil
|
|
81
|
+
}
|
|
@@ -0,0 +1,539 @@
|
|
|
1
|
+
package storage
|
|
2
|
+
|
|
3
|
+
import (
|
|
4
|
+
"context"
|
|
5
|
+
"crypto/sha256"
|
|
6
|
+
"encoding/binary"
|
|
7
|
+
"encoding/hex"
|
|
8
|
+
"encoding/json"
|
|
9
|
+
"errors"
|
|
10
|
+
"fmt"
|
|
11
|
+
"io"
|
|
12
|
+
"math"
|
|
13
|
+
"os"
|
|
14
|
+
"strings"
|
|
15
|
+
"time"
|
|
16
|
+
|
|
17
|
+
"github.com/aldanedev-create/rubydb/accelerator/internal/execution"
|
|
18
|
+
)
|
|
19
|
+
|
|
20
|
+
const (
|
|
21
|
+
snapshotFormatVersion = 1
|
|
22
|
+
pageHeaderSize = 64
|
|
23
|
+
recordHeaderSize = 16
|
|
24
|
+
nullBitmapFlag = 0x02
|
|
25
|
+
variablePrefixFlag = 0x04
|
|
26
|
+
deletedRecordFlag = 0x01
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
var (
|
|
30
|
+
ErrInvalidSnapshot = errors.New("invalid immutable storage snapshot")
|
|
31
|
+
ErrSnapshotSchema = errors.New("immutable snapshot schema is invalid")
|
|
32
|
+
)
|
|
33
|
+
|
|
34
|
+
type SnapshotManifest struct {
|
|
35
|
+
FormatVersion int `json:"format_version"`
|
|
36
|
+
SnapshotID string `json:"snapshot_id"`
|
|
37
|
+
SnapshotPath string `json:"snapshot_path"`
|
|
38
|
+
PageSize int `json:"page_size"`
|
|
39
|
+
PageCount uint64 `json:"page_count"`
|
|
40
|
+
FileSHA256 string `json:"file_sha256"`
|
|
41
|
+
Tables map[string]SnapshotTable `json:"tables"`
|
|
42
|
+
HiddenRowIDs []uint64 `json:"hidden_row_ids"`
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
type SnapshotTable struct {
|
|
46
|
+
Pages []uint64 `json:"pages"`
|
|
47
|
+
Columns []SnapshotColumn `json:"columns"`
|
|
48
|
+
Indexes []SnapshotIndex `json:"indexes"`
|
|
49
|
+
RowCount int `json:"row_count"`
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
type SnapshotColumn struct {
|
|
53
|
+
Name string `json:"name"`
|
|
54
|
+
Type string `json:"type"`
|
|
55
|
+
Nullable bool `json:"nullable"`
|
|
56
|
+
Default interface{} `json:"default,omitempty"`
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
type SnapshotIndex struct {
|
|
60
|
+
Name string `json:"name"`
|
|
61
|
+
Type string `json:"type"`
|
|
62
|
+
Columns []string `json:"columns"`
|
|
63
|
+
Unique bool `json:"unique"`
|
|
64
|
+
Entries []SnapshotEntry `json:"entries"`
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
type SnapshotEntry struct {
|
|
68
|
+
Key interface{} `json:"key"`
|
|
69
|
+
RowID uint64 `json:"row_id"`
|
|
70
|
+
}
|
|
71
|
+
|
|
72
|
+
type SnapshotScanRequest struct {
|
|
73
|
+
Snapshot SnapshotManifest `json:"snapshot"`
|
|
74
|
+
Table string `json:"table"`
|
|
75
|
+
Columns []string `json:"columns"`
|
|
76
|
+
ColumnTypes map[string]string `json:"column_types,omitempty"`
|
|
77
|
+
Filters []execution.Filter `json:"filters,omitempty"`
|
|
78
|
+
OrderBy []execution.Order `json:"order_by,omitempty"`
|
|
79
|
+
IndexName string `json:"index_name,omitempty"`
|
|
80
|
+
Limit *int `json:"limit,omitempty"`
|
|
81
|
+
Offset int `json:"offset,omitempty"`
|
|
82
|
+
Distinct bool `json:"distinct,omitempty"`
|
|
83
|
+
DistinctColumns []string `json:"distinct_columns,omitempty"`
|
|
84
|
+
BatchSize int `json:"batch_size,omitempty"`
|
|
85
|
+
MaxRows int `json:"max_rows,omitempty"`
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
type SnapshotScanResult struct {
|
|
89
|
+
Rows []map[string]interface{}
|
|
90
|
+
ColumnTypes map[string]string
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
// ExecuteSnapshotScan validates and reads only the file named by the
|
|
94
|
+
// lock-protected manifest. In direct mode that is the already-flushed live
|
|
95
|
+
// file; in detached mode it is a short-lived copy. It never consults WAL or
|
|
96
|
+
// MVCC state; Ruby has already established the eligible read snapshot.
|
|
97
|
+
func ExecuteSnapshotScan(request SnapshotScanRequest) (SnapshotScanResult, error) {
|
|
98
|
+
return ExecuteSnapshotScanContext(context.Background(), request)
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
// ExecuteSnapshotScanContext is cancellation-aware and stops page work as
|
|
102
|
+
// soon as the request context is cancelled. Ruby owns snapshot creation and
|
|
103
|
+
// visibility; Go only reads the file named by the manifest while Ruby holds
|
|
104
|
+
// the engine snapshot lock.
|
|
105
|
+
func ExecuteSnapshotScanContext(ctx context.Context, request SnapshotScanRequest) (SnapshotScanResult, error) {
|
|
106
|
+
if err := validateManifest(request.Snapshot); err != nil {
|
|
107
|
+
return SnapshotScanResult{}, err
|
|
108
|
+
}
|
|
109
|
+
file, err := os.Open(request.Snapshot.SnapshotPath)
|
|
110
|
+
if err != nil {
|
|
111
|
+
return SnapshotScanResult{}, fmt.Errorf("open snapshot: %w", err)
|
|
112
|
+
}
|
|
113
|
+
defer file.Close()
|
|
114
|
+
|
|
115
|
+
if err := validateSnapshotFile(file, request.Snapshot); err != nil {
|
|
116
|
+
return SnapshotScanResult{}, err
|
|
117
|
+
}
|
|
118
|
+
table, ok := findTable(request.Snapshot.Tables, request.Table)
|
|
119
|
+
if !ok || len(table.Columns) == 0 {
|
|
120
|
+
return SnapshotScanResult{}, fmt.Errorf("%w: table %q", ErrSnapshotSchema, request.Table)
|
|
121
|
+
}
|
|
122
|
+
columns, err := selectedColumns(table.Columns, request.Columns)
|
|
123
|
+
if err != nil {
|
|
124
|
+
return SnapshotScanResult{}, err
|
|
125
|
+
}
|
|
126
|
+
hidden := make(map[uint64]struct{}, len(request.Snapshot.HiddenRowIDs))
|
|
127
|
+
for _, rowID := range request.Snapshot.HiddenRowIDs {
|
|
128
|
+
hidden[rowID] = struct{}{}
|
|
129
|
+
}
|
|
130
|
+
indexed, useIndex, err := indexedRowIDs(table.Indexes, request)
|
|
131
|
+
if err != nil {
|
|
132
|
+
return SnapshotScanResult{}, err
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
rows := make([]map[string]interface{}, 0, table.RowCount)
|
|
136
|
+
needed := -1
|
|
137
|
+
if len(request.OrderBy) == 0 && !request.Distinct && request.Limit != nil {
|
|
138
|
+
if *request.Limit < 0 || request.Offset < 0 {
|
|
139
|
+
return SnapshotScanResult{}, execution.ErrInvalidWindow("limit and offset cannot be negative")
|
|
140
|
+
}
|
|
141
|
+
needed = request.Offset + *request.Limit
|
|
142
|
+
}
|
|
143
|
+
for _, pageNumber := range table.Pages {
|
|
144
|
+
select {
|
|
145
|
+
case <-ctx.Done():
|
|
146
|
+
return SnapshotScanResult{}, ctx.Err()
|
|
147
|
+
default:
|
|
148
|
+
}
|
|
149
|
+
page, err := readSnapshotPage(file, request.Snapshot, pageNumber)
|
|
150
|
+
if err != nil {
|
|
151
|
+
return SnapshotScanResult{}, err
|
|
152
|
+
}
|
|
153
|
+
pageRows, err := scanPage(page, table.Columns, columns, hidden, indexed, useIndex, request.Filters)
|
|
154
|
+
if err != nil {
|
|
155
|
+
return SnapshotScanResult{}, fmt.Errorf("scan page %d: %w", pageNumber, err)
|
|
156
|
+
}
|
|
157
|
+
rows = append(rows, pageRows...)
|
|
158
|
+
if request.MaxRows > 0 && len(rows) > request.MaxRows {
|
|
159
|
+
return SnapshotScanResult{}, fmt.Errorf("snapshot scan exceeds configured row limit")
|
|
160
|
+
}
|
|
161
|
+
if needed >= 0 && len(rows) >= needed {
|
|
162
|
+
break
|
|
163
|
+
}
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
if len(request.OrderBy) > 0 {
|
|
167
|
+
execution.SortRows(rows, request.OrderBy)
|
|
168
|
+
}
|
|
169
|
+
if request.Distinct {
|
|
170
|
+
columns := request.DistinctColumns
|
|
171
|
+
if len(columns) == 0 {
|
|
172
|
+
columns = request.Columns
|
|
173
|
+
}
|
|
174
|
+
rows = execution.DistinctRows(rows, columns)
|
|
175
|
+
}
|
|
176
|
+
rows = applyWindow(rows, request.Offset, request.Limit)
|
|
177
|
+
columnTypes := make(map[string]string, len(columns))
|
|
178
|
+
for _, column := range columns {
|
|
179
|
+
columnTypes[column.Name] = column.Type
|
|
180
|
+
}
|
|
181
|
+
return SnapshotScanResult{Rows: rows, ColumnTypes: columnTypes}, nil
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
func validateManifest(manifest SnapshotManifest) error {
|
|
185
|
+
if manifest.FormatVersion != snapshotFormatVersion || manifest.SnapshotID == "" || manifest.SnapshotPath == "" {
|
|
186
|
+
return ErrInvalidSnapshot
|
|
187
|
+
}
|
|
188
|
+
if manifest.PageSize < pageHeaderSize || manifest.PageSize > 64*1024*1024 || manifest.PageCount == 0 {
|
|
189
|
+
return ErrInvalidSnapshot
|
|
190
|
+
}
|
|
191
|
+
if len(manifest.Tables) == 0 {
|
|
192
|
+
return ErrInvalidSnapshot
|
|
193
|
+
}
|
|
194
|
+
if manifest.FileSHA256 != "" {
|
|
195
|
+
if len(manifest.FileSHA256) != sha256.Size*2 {
|
|
196
|
+
return ErrInvalidSnapshot
|
|
197
|
+
}
|
|
198
|
+
if _, err := hex.DecodeString(manifest.FileSHA256); err != nil {
|
|
199
|
+
return ErrInvalidSnapshot
|
|
200
|
+
}
|
|
201
|
+
}
|
|
202
|
+
return nil
|
|
203
|
+
}
|
|
204
|
+
|
|
205
|
+
func validateSnapshotFile(file *os.File, manifest SnapshotManifest) error {
|
|
206
|
+
info, err := file.Stat()
|
|
207
|
+
if err != nil || info.Size() != int64(manifest.PageSize)*int64(manifest.PageCount) {
|
|
208
|
+
return ErrInvalidSnapshot
|
|
209
|
+
}
|
|
210
|
+
if manifest.FileSHA256 != "" {
|
|
211
|
+
if _, err := file.Seek(0, io.SeekStart); err != nil {
|
|
212
|
+
return ErrInvalidSnapshot
|
|
213
|
+
}
|
|
214
|
+
digest := sha256.New()
|
|
215
|
+
if _, err := io.Copy(digest, file); err != nil || !strings.EqualFold(hex.EncodeToString(digest.Sum(nil)), manifest.FileSHA256) {
|
|
216
|
+
return ErrInvalidSnapshot
|
|
217
|
+
}
|
|
218
|
+
}
|
|
219
|
+
page, err := readSnapshotPage(file, manifest, 0)
|
|
220
|
+
if err != nil || binary.BigEndian.Uint64(page[0:8]) != 0 || binary.BigEndian.Uint64(page[8:16]) != uint64(manifest.PageSize) ||
|
|
221
|
+
binary.BigEndian.Uint32(page[16:20]) != pageHeaderSize || binary.BigEndian.Uint32(page[32:36]) != 1 {
|
|
222
|
+
return ErrInvalidSnapshot
|
|
223
|
+
}
|
|
224
|
+
return nil
|
|
225
|
+
}
|
|
226
|
+
|
|
227
|
+
func readSnapshotPage(file *os.File, manifest SnapshotManifest, pageNumber uint64) ([]byte, error) {
|
|
228
|
+
if pageNumber >= manifest.PageCount {
|
|
229
|
+
return nil, ErrInvalidSnapshot
|
|
230
|
+
}
|
|
231
|
+
page := make([]byte, manifest.PageSize)
|
|
232
|
+
read, err := file.ReadAt(page, int64(pageNumber)*int64(manifest.PageSize))
|
|
233
|
+
if read != len(page) || (err != nil && !errors.Is(err, io.EOF)) {
|
|
234
|
+
return nil, ErrInvalidSnapshot
|
|
235
|
+
}
|
|
236
|
+
if binary.BigEndian.Uint64(page[0:8]) != pageNumber || binary.BigEndian.Uint64(page[8:16]) != uint64(manifest.PageSize) {
|
|
237
|
+
return nil, ErrInvalidSnapshot
|
|
238
|
+
}
|
|
239
|
+
dataEnd := binary.BigEndian.Uint32(page[20:24])
|
|
240
|
+
if dataEnd < pageHeaderSize || dataEnd > uint32(len(page)) {
|
|
241
|
+
return nil, ErrInvalidSnapshot
|
|
242
|
+
}
|
|
243
|
+
return page, nil
|
|
244
|
+
}
|
|
245
|
+
|
|
246
|
+
func findTable(tables map[string]SnapshotTable, name string) (SnapshotTable, bool) {
|
|
247
|
+
if table, ok := tables[name]; ok {
|
|
248
|
+
return table, true
|
|
249
|
+
}
|
|
250
|
+
for tableName, table := range tables {
|
|
251
|
+
if strings.EqualFold(tableName, name) {
|
|
252
|
+
return table, true
|
|
253
|
+
}
|
|
254
|
+
}
|
|
255
|
+
return SnapshotTable{}, false
|
|
256
|
+
}
|
|
257
|
+
|
|
258
|
+
func selectedColumns(all []SnapshotColumn, requested []string) ([]SnapshotColumn, error) {
|
|
259
|
+
if len(requested) == 0 {
|
|
260
|
+
return all, nil
|
|
261
|
+
}
|
|
262
|
+
byName := make(map[string]SnapshotColumn, len(all))
|
|
263
|
+
for _, column := range all {
|
|
264
|
+
byName[strings.ToLower(column.Name)] = column
|
|
265
|
+
}
|
|
266
|
+
selected := make([]SnapshotColumn, 0, len(requested))
|
|
267
|
+
for _, name := range requested {
|
|
268
|
+
column, ok := byName[strings.ToLower(name)]
|
|
269
|
+
if !ok {
|
|
270
|
+
return nil, fmt.Errorf("%w: unknown column %q", ErrSnapshotSchema, name)
|
|
271
|
+
}
|
|
272
|
+
selected = append(selected, column)
|
|
273
|
+
}
|
|
274
|
+
return selected, nil
|
|
275
|
+
}
|
|
276
|
+
|
|
277
|
+
func scanPage(page []byte, schema, selected []SnapshotColumn, hidden, indexed map[uint64]struct{}, useIndex bool, filters []execution.Filter) ([]map[string]interface{}, error) {
|
|
278
|
+
dataEnd := int(binary.BigEndian.Uint32(page[20:24]))
|
|
279
|
+
rows := make([]map[string]interface{}, 0)
|
|
280
|
+
for offset := pageHeaderSize; offset < dataEnd; {
|
|
281
|
+
if dataEnd-offset < recordHeaderSize {
|
|
282
|
+
return nil, fmt.Errorf("record header at offset %d: %w", offset, ErrInvalidSnapshot)
|
|
283
|
+
}
|
|
284
|
+
rowID := binary.BigEndian.Uint64(page[offset : offset+8])
|
|
285
|
+
recordSize := int(binary.BigEndian.Uint32(page[offset+8 : offset+12]))
|
|
286
|
+
// Ruby's record header is Q>L>S>S: the flags field is explicitly
|
|
287
|
+
// big-endian while the final column-count S is native-endian.
|
|
288
|
+
flags := binary.BigEndian.Uint16(page[offset+12 : offset+14])
|
|
289
|
+
columnCount := int(binary.LittleEndian.Uint16(page[offset+14 : offset+16]))
|
|
290
|
+
offset += recordHeaderSize
|
|
291
|
+
if recordSize < 0 || offset+recordSize > dataEnd || columnCount < 0 || columnCount > len(schema) {
|
|
292
|
+
return nil, fmt.Errorf("record at offset %d has size %d and %d columns: %w", offset-recordHeaderSize, recordSize, columnCount, ErrInvalidSnapshot)
|
|
293
|
+
}
|
|
294
|
+
record := page[offset : offset+recordSize]
|
|
295
|
+
offset += recordSize
|
|
296
|
+
if flags&deletedRecordFlag != 0 {
|
|
297
|
+
continue
|
|
298
|
+
}
|
|
299
|
+
if _, ok := hidden[rowID]; ok {
|
|
300
|
+
continue
|
|
301
|
+
}
|
|
302
|
+
if useIndex {
|
|
303
|
+
if _, ok := indexed[rowID]; !ok {
|
|
304
|
+
continue
|
|
305
|
+
}
|
|
306
|
+
}
|
|
307
|
+
row, err := decodeRow(record, flags, columnCount, schema, selected)
|
|
308
|
+
if err != nil {
|
|
309
|
+
return nil, fmt.Errorf("row %d: %w", rowID, err)
|
|
310
|
+
}
|
|
311
|
+
row["_row_id"] = rowID
|
|
312
|
+
if !matchesFilters(row, filters) {
|
|
313
|
+
continue
|
|
314
|
+
}
|
|
315
|
+
rows = append(rows, row)
|
|
316
|
+
}
|
|
317
|
+
return rows, nil
|
|
318
|
+
}
|
|
319
|
+
|
|
320
|
+
func decodeRow(data []byte, flags uint16, columnCount int, schema, selected []SnapshotColumn) (map[string]interface{}, error) {
|
|
321
|
+
bitmapSize := 0
|
|
322
|
+
if flags&nullBitmapFlag != 0 {
|
|
323
|
+
bitmapSize = (len(schema) + 7) / 8
|
|
324
|
+
if bitmapSize > len(data) {
|
|
325
|
+
return nil, ErrInvalidSnapshot
|
|
326
|
+
}
|
|
327
|
+
}
|
|
328
|
+
bitmap := data[:bitmapSize]
|
|
329
|
+
offset := bitmapSize
|
|
330
|
+
selectedNames := make(map[string]struct{}, len(selected))
|
|
331
|
+
for _, column := range selected {
|
|
332
|
+
selectedNames[strings.ToLower(column.Name)] = struct{}{}
|
|
333
|
+
}
|
|
334
|
+
row := make(map[string]interface{}, len(selected))
|
|
335
|
+
for index := 0; index < len(schema); index++ {
|
|
336
|
+
column := schema[index]
|
|
337
|
+
if index >= columnCount {
|
|
338
|
+
if _, ok := selectedNames[strings.ToLower(column.Name)]; ok {
|
|
339
|
+
row[column.Name] = column.Default
|
|
340
|
+
}
|
|
341
|
+
continue
|
|
342
|
+
}
|
|
343
|
+
value, next, err := decodeColumn(data, offset, column.Type, flags&variablePrefixFlag != 0)
|
|
344
|
+
if err != nil {
|
|
345
|
+
return nil, fmt.Errorf("column %q: %w", column.Name, err)
|
|
346
|
+
}
|
|
347
|
+
offset = next
|
|
348
|
+
if bitmapSet(bitmap, index) {
|
|
349
|
+
value = nil
|
|
350
|
+
}
|
|
351
|
+
if value == nil && column.Default != nil {
|
|
352
|
+
value = column.Default
|
|
353
|
+
}
|
|
354
|
+
if _, ok := selectedNames[strings.ToLower(column.Name)]; ok {
|
|
355
|
+
row[column.Name] = value
|
|
356
|
+
}
|
|
357
|
+
}
|
|
358
|
+
if offset > len(data) {
|
|
359
|
+
return nil, ErrInvalidSnapshot
|
|
360
|
+
}
|
|
361
|
+
return row, nil
|
|
362
|
+
}
|
|
363
|
+
|
|
364
|
+
func decodeColumn(data []byte, offset int, typeName string, variablePrefixes bool) (interface{}, int, error) {
|
|
365
|
+
typeName = strings.ToLower(typeName)
|
|
366
|
+
size := fixedSize(typeName)
|
|
367
|
+
if size > 0 {
|
|
368
|
+
if offset+size > len(data) {
|
|
369
|
+
return nil, offset, ErrInvalidSnapshot
|
|
370
|
+
}
|
|
371
|
+
raw := data[offset : offset+size]
|
|
372
|
+
return decodeFixed(raw, typeName), offset + size, nil
|
|
373
|
+
}
|
|
374
|
+
if !variablePrefixes {
|
|
375
|
+
return nil, len(data), nil
|
|
376
|
+
}
|
|
377
|
+
if offset+4 > len(data) {
|
|
378
|
+
return nil, offset, ErrInvalidSnapshot
|
|
379
|
+
}
|
|
380
|
+
length := int(binary.BigEndian.Uint32(data[offset : offset+4]))
|
|
381
|
+
offset += 4
|
|
382
|
+
if length < 0 || offset+length > len(data) {
|
|
383
|
+
return nil, offset, ErrInvalidSnapshot
|
|
384
|
+
}
|
|
385
|
+
return decodeVariable(data[offset:offset+length], typeName), offset + length, nil
|
|
386
|
+
}
|
|
387
|
+
|
|
388
|
+
func fixedSize(typeName string) int {
|
|
389
|
+
switch typeName {
|
|
390
|
+
case "integer", "date":
|
|
391
|
+
return 4
|
|
392
|
+
case "smallint":
|
|
393
|
+
return 2
|
|
394
|
+
case "bigint", "float", "time", "timestamp":
|
|
395
|
+
return 8
|
|
396
|
+
case "boolean":
|
|
397
|
+
return 1
|
|
398
|
+
case "uuid":
|
|
399
|
+
return 16
|
|
400
|
+
default:
|
|
401
|
+
return 0
|
|
402
|
+
}
|
|
403
|
+
}
|
|
404
|
+
|
|
405
|
+
func decodeFixed(raw []byte, typeName string) interface{} {
|
|
406
|
+
switch typeName {
|
|
407
|
+
case "integer":
|
|
408
|
+
return int64(int32(binary.BigEndian.Uint32(raw)))
|
|
409
|
+
case "smallint":
|
|
410
|
+
return int64(int16(binary.BigEndian.Uint16(raw)))
|
|
411
|
+
case "bigint":
|
|
412
|
+
return int64(binary.BigEndian.Uint64(raw))
|
|
413
|
+
case "float":
|
|
414
|
+
return math.Float64frombits(binary.BigEndian.Uint64(raw))
|
|
415
|
+
case "boolean":
|
|
416
|
+
return raw[0] == 1
|
|
417
|
+
case "date":
|
|
418
|
+
return time.Unix(int64(int32(binary.BigEndian.Uint32(raw)))*86400, 0).UTC().Format("2006-01-02")
|
|
419
|
+
case "time":
|
|
420
|
+
total := int64(binary.BigEndian.Uint64(raw))
|
|
421
|
+
seconds, micros := total/1_000_000, total%1_000_000
|
|
422
|
+
value := time.Unix(0, 0).UTC().Add(time.Duration(seconds)*time.Second + time.Duration(micros)*time.Microsecond)
|
|
423
|
+
return value.Format(time.RFC3339Nano)
|
|
424
|
+
case "timestamp":
|
|
425
|
+
return time.Unix(int64(binary.BigEndian.Uint64(raw)), 0).UTC().Format(time.RFC3339Nano)
|
|
426
|
+
case "uuid":
|
|
427
|
+
encoded := hex.EncodeToString(raw)
|
|
428
|
+
return encoded[0:8] + "-" + encoded[8:12] + "-" + encoded[12:16] + "-" + encoded[16:20] + "-" + encoded[20:32]
|
|
429
|
+
default:
|
|
430
|
+
return string(raw)
|
|
431
|
+
}
|
|
432
|
+
}
|
|
433
|
+
|
|
434
|
+
func decodeVariable(raw []byte, typeName string) interface{} {
|
|
435
|
+
if typeName == "blob" {
|
|
436
|
+
return append([]byte(nil), raw...)
|
|
437
|
+
}
|
|
438
|
+
if typeName == "json" {
|
|
439
|
+
var value interface{}
|
|
440
|
+
if json.Unmarshal(raw, &value) == nil {
|
|
441
|
+
return value
|
|
442
|
+
}
|
|
443
|
+
}
|
|
444
|
+
return string(raw)
|
|
445
|
+
}
|
|
446
|
+
|
|
447
|
+
func bitmapSet(bitmap []byte, index int) bool {
|
|
448
|
+
return len(bitmap) > index/8 && bitmap[index/8]&(1<<uint(index%8)) != 0
|
|
449
|
+
}
|
|
450
|
+
|
|
451
|
+
func matchesFilters(row map[string]interface{}, filters []execution.Filter) bool {
|
|
452
|
+
for _, filter := range filters {
|
|
453
|
+
column := filter.Column
|
|
454
|
+
if dot := strings.LastIndex(column, "."); dot >= 0 {
|
|
455
|
+
column = column[dot+1:]
|
|
456
|
+
}
|
|
457
|
+
if !execution.Matches(row[column], filter.Operator, filter.Value) {
|
|
458
|
+
return false
|
|
459
|
+
}
|
|
460
|
+
}
|
|
461
|
+
return true
|
|
462
|
+
}
|
|
463
|
+
|
|
464
|
+
func indexedRowIDs(indexes []SnapshotIndex, request SnapshotScanRequest) (map[uint64]struct{}, bool, error) {
|
|
465
|
+
if request.IndexName == "" {
|
|
466
|
+
return nil, false, nil
|
|
467
|
+
}
|
|
468
|
+
var selected *SnapshotIndex
|
|
469
|
+
for index := range indexes {
|
|
470
|
+
if strings.EqualFold(indexes[index].Name, request.IndexName) {
|
|
471
|
+
selected = &indexes[index]
|
|
472
|
+
break
|
|
473
|
+
}
|
|
474
|
+
}
|
|
475
|
+
if selected == nil || !strings.EqualFold(selected.Type, "btree") {
|
|
476
|
+
return nil, false, nil
|
|
477
|
+
}
|
|
478
|
+
rowIDs := make(map[uint64]struct{})
|
|
479
|
+
matchedFilter := false
|
|
480
|
+
for _, entry := range selected.Entries {
|
|
481
|
+
matches := true
|
|
482
|
+
for index, column := range selected.Columns {
|
|
483
|
+
key, ok := indexKeyPart(entry.Key, index, len(selected.Columns))
|
|
484
|
+
if !ok {
|
|
485
|
+
matches = false
|
|
486
|
+
break
|
|
487
|
+
}
|
|
488
|
+
for _, filter := range request.Filters {
|
|
489
|
+
filterColumn := filter.Column
|
|
490
|
+
if dot := strings.LastIndex(filterColumn, "."); dot >= 0 {
|
|
491
|
+
filterColumn = filterColumn[dot+1:]
|
|
492
|
+
}
|
|
493
|
+
if strings.EqualFold(filterColumn, column) {
|
|
494
|
+
matchedFilter = true
|
|
495
|
+
if !execution.Matches(key, filter.Operator, filter.Value) {
|
|
496
|
+
matches = false
|
|
497
|
+
}
|
|
498
|
+
}
|
|
499
|
+
}
|
|
500
|
+
}
|
|
501
|
+
if matches {
|
|
502
|
+
rowIDs[entry.RowID] = struct{}{}
|
|
503
|
+
}
|
|
504
|
+
}
|
|
505
|
+
if !matchedFilter {
|
|
506
|
+
return nil, false, nil
|
|
507
|
+
}
|
|
508
|
+
return rowIDs, true, nil
|
|
509
|
+
}
|
|
510
|
+
|
|
511
|
+
func indexKeyPart(key interface{}, index, count int) (interface{}, bool) {
|
|
512
|
+
if count == 1 {
|
|
513
|
+
return key, true
|
|
514
|
+
}
|
|
515
|
+
parts, ok := key.([]interface{})
|
|
516
|
+
if !ok || index >= len(parts) {
|
|
517
|
+
return nil, false
|
|
518
|
+
}
|
|
519
|
+
return parts[index], true
|
|
520
|
+
}
|
|
521
|
+
|
|
522
|
+
func applyWindow(rows []map[string]interface{}, offset int, limit *int) []map[string]interface{} {
|
|
523
|
+
if offset < 0 {
|
|
524
|
+
offset = 0
|
|
525
|
+
}
|
|
526
|
+
if offset >= len(rows) {
|
|
527
|
+
return []map[string]interface{}{}
|
|
528
|
+
}
|
|
529
|
+
rows = rows[offset:]
|
|
530
|
+
if limit != nil {
|
|
531
|
+
if *limit <= 0 {
|
|
532
|
+
return []map[string]interface{}{}
|
|
533
|
+
}
|
|
534
|
+
if *limit < len(rows) {
|
|
535
|
+
rows = rows[:*limit]
|
|
536
|
+
}
|
|
537
|
+
}
|
|
538
|
+
return rows
|
|
539
|
+
}
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
package wal
|
|
2
|
+
|
|
3
|
+
import (
|
|
4
|
+
"bytes"
|
|
5
|
+
"compress/gzip"
|
|
6
|
+
"io"
|
|
7
|
+
)
|
|
8
|
+
|
|
9
|
+
func Gzip(data []byte, level int) ([]byte, error) {
|
|
10
|
+
var output bytes.Buffer
|
|
11
|
+
writer, err := gzip.NewWriterLevel(&output, level)
|
|
12
|
+
if err != nil {
|
|
13
|
+
return nil, err
|
|
14
|
+
}
|
|
15
|
+
if _, err = writer.Write(data); err == nil {
|
|
16
|
+
err = writer.Close()
|
|
17
|
+
}
|
|
18
|
+
if err != nil {
|
|
19
|
+
return nil, err
|
|
20
|
+
}
|
|
21
|
+
return output.Bytes(), nil
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
func Gunzip(data []byte, maxOutput int) ([]byte, error) {
|
|
25
|
+
reader, err := gzip.NewReader(bytes.NewReader(data))
|
|
26
|
+
if err != nil {
|
|
27
|
+
return nil, err
|
|
28
|
+
}
|
|
29
|
+
output, readErr := io.ReadAll(io.LimitReader(reader, int64(maxOutput)+1))
|
|
30
|
+
closeErr := reader.Close()
|
|
31
|
+
if readErr != nil {
|
|
32
|
+
return nil, readErr
|
|
33
|
+
}
|
|
34
|
+
if len(output) > maxOutput {
|
|
35
|
+
return nil, io.ErrShortBuffer
|
|
36
|
+
}
|
|
37
|
+
if closeErr != nil {
|
|
38
|
+
return nil, closeErr
|
|
39
|
+
}
|
|
40
|
+
return output, nil
|
|
41
|
+
}
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
package wal
|
|
2
|
+
|
|
3
|
+
import "sync"
|
|
4
|
+
|
|
5
|
+
// Batch collects approved WAL payloads before Ruby assigns the durable LSN.
|
|
6
|
+
// It does not perform fsync or publish commits; those remain RubyDB rules.
|
|
7
|
+
type Batch struct {
|
|
8
|
+
mu sync.Mutex
|
|
9
|
+
payload [][]byte
|
|
10
|
+
}
|
|
11
|
+
|
|
12
|
+
func (batch *Batch) Add(payload []byte) {
|
|
13
|
+
batch.mu.Lock()
|
|
14
|
+
defer batch.mu.Unlock()
|
|
15
|
+
batch.payload = append(batch.payload, append([]byte(nil), payload...))
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
func (batch *Batch) Drain() [][]byte {
|
|
19
|
+
batch.mu.Lock()
|
|
20
|
+
defer batch.mu.Unlock()
|
|
21
|
+
payload := batch.payload
|
|
22
|
+
batch.payload = nil
|
|
23
|
+
return payload
|
|
24
|
+
}
|