rubydb 0.1.5 → 0.1.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.gitignore +4 -0
- data/CHANGELOG.md +14 -0
- data/Gemfile.lock +1 -1
- data/README.md +296 -227
- data/Rakefile +6 -1
- data/accelerator/bin/SHA256SUMS +6 -0
- data/accelerator/bin/rubydb-accelerator-darwin-amd64 +0 -0
- data/accelerator/bin/rubydb-accelerator-darwin-arm64 +0 -0
- data/accelerator/bin/rubydb-accelerator-linux-amd64 +0 -0
- data/accelerator/bin/rubydb-accelerator-linux-arm64 +0 -0
- data/accelerator/bin/rubydb-accelerator-windows-amd64.exe +0 -0
- data/accelerator/bin/rubydb-accelerator-windows-arm64.exe +0 -0
- data/accelerator/cmd/rubydb-accelerator/main.go +11 -0
- data/accelerator/go.mod +3 -0
- data/accelerator/internal/execution/aggregate.go +94 -0
- data/accelerator/internal/execution/distinct.go +22 -0
- data/accelerator/internal/execution/filter.go +73 -0
- data/accelerator/internal/execution/join.go +79 -0
- data/accelerator/internal/execution/operators.go +167 -0
- data/accelerator/internal/execution/scan.go +20 -0
- data/accelerator/internal/execution/sort.go +62 -0
- data/accelerator/internal/execution/types.go +136 -0
- data/accelerator/internal/execution/value.go +67 -0
- data/accelerator/internal/memory/arena.go +47 -0
- data/accelerator/internal/memory/reuse.go +22 -0
- data/accelerator/internal/metrics/registry.go +67 -0
- data/accelerator/internal/parallel/bounded_queue.go +56 -0
- data/accelerator/internal/parallel/scheduler.go +47 -0
- data/accelerator/internal/parallel/worker_pool.go +53 -0
- data/accelerator/internal/protocol/cancellation.go +48 -0
- data/accelerator/internal/protocol/columnar.go +263 -0
- data/accelerator/internal/protocol/frame.go +187 -0
- data/accelerator/internal/runtime/worker.go +521 -0
- data/accelerator/internal/storage/page_reader.go +81 -0
- data/accelerator/internal/storage/snapshot_scan.go +539 -0
- data/accelerator/internal/wal/checksum.go +13 -0
- data/accelerator/internal/wal/compression.go +41 -0
- data/accelerator/internal/wal/group_commit.go +24 -0
- data/accelerator/internal/wal/record_encoder.go +40 -0
- data/adapters/activerecord/README.md +8 -3
- data/adapters/activerecord/lib/active_record/connection_adapters/rubydb_adapter.rb +50 -42
- data/adapters/activerecord/rubydb-activerecord.gemspec +1 -1
- data/docs/README.md +3 -1
- data/docs/architecture/go-accelerator.md +179 -0
- data/docs/cli.md +24 -0
- data/docs/contributing/benchmarking.md +16 -0
- data/docs/developer/local-development.md +32 -0
- data/docs/release.md +2 -2
- data/lessons/02-local-development.md +2 -2
- data/lessons/04-rails-complex-apps.md +2 -2
- data/lessons/05-rubydb-production-server.md +2 -2
- data/lessons/07-hybrid-microservices.md +175 -90
- data/lessons/10-release-readiness.md +184 -117
- data/lessons/11-community-adapter.md +323 -0
- data/lessons/12-rails-ecommerce-pressure.md +263 -0
- data/lib/rubydb/accelerator/client.rb +451 -0
- data/lib/rubydb/accelerator/error.rb +22 -0
- data/lib/rubydb/accelerator/manager.rb +606 -0
- data/lib/rubydb/accelerator.rb +13 -0
- data/lib/rubydb/cli/application.rb +6 -1
- data/lib/rubydb/cli/commands/accelerator.rb +72 -0
- data/lib/rubydb/cli/commands/doctor.rb +3 -0
- data/lib/rubydb/client/client.rb +7 -0
- data/lib/rubydb/client/connection.rb +15 -0
- data/lib/rubydb/client/result.rb +5 -1
- data/lib/rubydb/configuration/defaults.rb +12 -0
- data/lib/rubydb/configuration/validation.rb +8 -1
- data/lib/rubydb/execution/accelerator_dispatch.rb +30 -0
- data/lib/rubydb/execution/cost_model.rb +72 -0
- data/lib/rubydb/execution/executor.rb +373 -11
- data/lib/rubydb/execution/operator_selection.rb +57 -0
- data/lib/rubydb/execution/physical_plan.rb +47 -0
- data/lib/rubydb/execution/planner.rb +12 -46
- data/lib/rubydb/execution/sort_executor.rb +22 -8
- data/lib/rubydb/indexes/btree.rb +31 -2
- data/lib/rubydb/rubydb.rb +7 -1
- data/lib/rubydb/server/session.rb +45 -0
- data/lib/rubydb/storage/engine.rb +74 -12
- data/lib/rubydb/storage/snapshot_reader.rb +167 -0
- data/lib/rubydb/version.rb +1 -1
- data/lib/rubydb/wal/archive.rb +17 -0
- data/lib/rubydb/wal/wal.rb +1 -0
- data/rubydb.gemspec +12 -2
- data/scripts/build_accelerator +49 -0
- data/scripts/release +34 -4
- data/scripts/replication_failover_drill +2 -2
- metadata +49 -1
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
package execution
|
|
2
|
+
|
|
3
|
+
import (
|
|
4
|
+
"encoding/json"
|
|
5
|
+
"fmt"
|
|
6
|
+
"math"
|
|
7
|
+
"strings"
|
|
8
|
+
)
|
|
9
|
+
|
|
10
|
+
func Compare(left, right interface{}) int {
|
|
11
|
+
if left == nil && right == nil {
|
|
12
|
+
return 0
|
|
13
|
+
}
|
|
14
|
+
if left == nil {
|
|
15
|
+
return -1
|
|
16
|
+
}
|
|
17
|
+
if right == nil {
|
|
18
|
+
return 1
|
|
19
|
+
}
|
|
20
|
+
if leftNumber, ok := NumberValue(left); ok {
|
|
21
|
+
if rightNumber, rightOK := NumberValue(right); rightOK {
|
|
22
|
+
switch {
|
|
23
|
+
case leftNumber < rightNumber:
|
|
24
|
+
return -1
|
|
25
|
+
case leftNumber > rightNumber:
|
|
26
|
+
return 1
|
|
27
|
+
default:
|
|
28
|
+
return 0
|
|
29
|
+
}
|
|
30
|
+
}
|
|
31
|
+
}
|
|
32
|
+
return strings.Compare(fmt.Sprint(left), fmt.Sprint(right))
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
func NumberValue(value interface{}) (float64, bool) {
|
|
36
|
+
switch number := value.(type) {
|
|
37
|
+
case json.Number:
|
|
38
|
+
parsed, err := number.Float64()
|
|
39
|
+
return parsed, err == nil
|
|
40
|
+
case float32:
|
|
41
|
+
return float64(number), !math.IsNaN(float64(number))
|
|
42
|
+
case float64:
|
|
43
|
+
return number, !math.IsNaN(number)
|
|
44
|
+
case int:
|
|
45
|
+
return float64(number), true
|
|
46
|
+
case int8:
|
|
47
|
+
return float64(number), true
|
|
48
|
+
case int16:
|
|
49
|
+
return float64(number), true
|
|
50
|
+
case int32:
|
|
51
|
+
return float64(number), true
|
|
52
|
+
case int64:
|
|
53
|
+
return float64(number), true
|
|
54
|
+
case uint:
|
|
55
|
+
return float64(number), true
|
|
56
|
+
case uint8:
|
|
57
|
+
return float64(number), true
|
|
58
|
+
case uint16:
|
|
59
|
+
return float64(number), true
|
|
60
|
+
case uint32:
|
|
61
|
+
return float64(number), true
|
|
62
|
+
case uint64:
|
|
63
|
+
return float64(number), true
|
|
64
|
+
default:
|
|
65
|
+
return 0, false
|
|
66
|
+
}
|
|
67
|
+
}
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
package memory
|
|
2
|
+
|
|
3
|
+
type Arena struct {
|
|
4
|
+
blocks [][]byte
|
|
5
|
+
used int
|
|
6
|
+
}
|
|
7
|
+
|
|
8
|
+
func NewArena(capacity int) *Arena {
|
|
9
|
+
if capacity < 1024 {
|
|
10
|
+
capacity = 1024
|
|
11
|
+
}
|
|
12
|
+
return &Arena{blocks: [][]byte{make([]byte, capacity)}}
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
func (arena *Arena) Bytes(size int) []byte {
|
|
16
|
+
if size < 0 {
|
|
17
|
+
size = 0
|
|
18
|
+
}
|
|
19
|
+
block := arena.blocks[len(arena.blocks)-1]
|
|
20
|
+
if arena.used+size > len(block) {
|
|
21
|
+
capacity := len(block) * 2
|
|
22
|
+
if capacity < size {
|
|
23
|
+
capacity = size
|
|
24
|
+
}
|
|
25
|
+
block = make([]byte, capacity)
|
|
26
|
+
arena.blocks = append(arena.blocks, block)
|
|
27
|
+
arena.used = 0
|
|
28
|
+
}
|
|
29
|
+
start := arena.used
|
|
30
|
+
arena.used += size
|
|
31
|
+
return block[start:arena.used]
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
func (arena *Arena) Reset() {
|
|
35
|
+
if len(arena.blocks) > 1 {
|
|
36
|
+
arena.blocks = arena.blocks[:1]
|
|
37
|
+
}
|
|
38
|
+
arena.used = 0
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
func (arena *Arena) Capacity() int {
|
|
42
|
+
total := 0
|
|
43
|
+
for _, block := range arena.blocks {
|
|
44
|
+
total += len(block)
|
|
45
|
+
}
|
|
46
|
+
return total
|
|
47
|
+
}
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
package memory
|
|
2
|
+
|
|
3
|
+
type BytesPool struct {
|
|
4
|
+
pool [][]byte
|
|
5
|
+
}
|
|
6
|
+
|
|
7
|
+
func (pool *BytesPool) Get(size int) []byte {
|
|
8
|
+
for index, candidate := range pool.pool {
|
|
9
|
+
if cap(candidate) >= size {
|
|
10
|
+
pool.pool = append(pool.pool[:index], pool.pool[index+1:]...)
|
|
11
|
+
return candidate[:size]
|
|
12
|
+
}
|
|
13
|
+
}
|
|
14
|
+
return make([]byte, size)
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
func (pool *BytesPool) Put(buffer []byte) {
|
|
18
|
+
if buffer == nil {
|
|
19
|
+
return
|
|
20
|
+
}
|
|
21
|
+
pool.pool = append(pool.pool, buffer[:0])
|
|
22
|
+
}
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
package metrics
|
|
2
|
+
|
|
3
|
+
import (
|
|
4
|
+
"sync"
|
|
5
|
+
"time"
|
|
6
|
+
)
|
|
7
|
+
|
|
8
|
+
type Sample struct {
|
|
9
|
+
Count int64 `json:"count"`
|
|
10
|
+
Total float64 `json:"total_ms"`
|
|
11
|
+
P50 float64 `json:"p50_ms"`
|
|
12
|
+
P95 float64 `json:"p95_ms"`
|
|
13
|
+
P99 float64 `json:"p99_ms"`
|
|
14
|
+
}
|
|
15
|
+
|
|
16
|
+
type Registry struct {
|
|
17
|
+
mu sync.Mutex
|
|
18
|
+
counts map[string]int64
|
|
19
|
+
totals map[string]float64
|
|
20
|
+
samples map[string][]float64
|
|
21
|
+
}
|
|
22
|
+
|
|
23
|
+
func NewRegistry() *Registry {
|
|
24
|
+
return &Registry{counts: map[string]int64{}, totals: map[string]float64{}, samples: map[string][]float64{}}
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
func (registry *Registry) Observe(operation string, duration time.Duration) {
|
|
28
|
+
registry.mu.Lock()
|
|
29
|
+
defer registry.mu.Unlock()
|
|
30
|
+
ms := float64(duration) / float64(time.Millisecond)
|
|
31
|
+
registry.counts[operation]++
|
|
32
|
+
registry.totals[operation] += ms
|
|
33
|
+
values := registry.samples[operation]
|
|
34
|
+
if len(values) >= 1024 {
|
|
35
|
+
copy(values, values[1:])
|
|
36
|
+
values = values[:1023]
|
|
37
|
+
}
|
|
38
|
+
registry.samples[operation] = append(values, ms)
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
func (registry *Registry) Snapshot() map[string]Sample {
|
|
42
|
+
registry.mu.Lock()
|
|
43
|
+
defer registry.mu.Unlock()
|
|
44
|
+
result := make(map[string]Sample, len(registry.counts))
|
|
45
|
+
for operation, count := range registry.counts {
|
|
46
|
+
values := append([]float64(nil), registry.samples[operation]...)
|
|
47
|
+
for index := 1; index < len(values); index++ {
|
|
48
|
+
value := values[index]
|
|
49
|
+
position := index - 1
|
|
50
|
+
for position >= 0 && values[position] > value {
|
|
51
|
+
values[position+1] = values[position]
|
|
52
|
+
position--
|
|
53
|
+
}
|
|
54
|
+
values[position+1] = value
|
|
55
|
+
}
|
|
56
|
+
p50, p95, p99 := float64(0), float64(0), float64(0)
|
|
57
|
+
if len(values) > 0 {
|
|
58
|
+
p50 = values[int(float64(len(values)-1)*0.50)]
|
|
59
|
+
position := int(float64(len(values)-1) * 0.95)
|
|
60
|
+
p95 = values[position]
|
|
61
|
+
position = int(float64(len(values)-1) * 0.99)
|
|
62
|
+
p99 = values[position]
|
|
63
|
+
}
|
|
64
|
+
result[operation] = Sample{Count: count, Total: registry.totals[operation], P50: p50, P95: p95, P99: p99}
|
|
65
|
+
}
|
|
66
|
+
return result
|
|
67
|
+
}
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
package parallel
|
|
2
|
+
|
|
3
|
+
import (
|
|
4
|
+
"context"
|
|
5
|
+
"errors"
|
|
6
|
+
)
|
|
7
|
+
|
|
8
|
+
var ErrQueueClosed = errors.New("bounded queue is closed")
|
|
9
|
+
|
|
10
|
+
type BoundedQueue[T any] struct {
|
|
11
|
+
items chan T
|
|
12
|
+
closed chan struct{}
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
func NewBoundedQueue[T any](capacity int) *BoundedQueue[T] {
|
|
16
|
+
if capacity < 1 {
|
|
17
|
+
capacity = 1
|
|
18
|
+
}
|
|
19
|
+
return &BoundedQueue[T]{items: make(chan T, capacity), closed: make(chan struct{})}
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
func (queue *BoundedQueue[T]) Push(ctx context.Context, item T) error {
|
|
23
|
+
select {
|
|
24
|
+
case <-queue.closed:
|
|
25
|
+
return ErrQueueClosed
|
|
26
|
+
case <-ctx.Done():
|
|
27
|
+
return ctx.Err()
|
|
28
|
+
case queue.items <- item:
|
|
29
|
+
return nil
|
|
30
|
+
}
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
func (queue *BoundedQueue[T]) Pop(ctx context.Context) (T, error) {
|
|
34
|
+
var zero T
|
|
35
|
+
select {
|
|
36
|
+
case item := <-queue.items:
|
|
37
|
+
return item, nil
|
|
38
|
+
case <-queue.closed:
|
|
39
|
+
select {
|
|
40
|
+
case item := <-queue.items:
|
|
41
|
+
return item, nil
|
|
42
|
+
default:
|
|
43
|
+
return zero, ErrQueueClosed
|
|
44
|
+
}
|
|
45
|
+
case <-ctx.Done():
|
|
46
|
+
return zero, ctx.Err()
|
|
47
|
+
}
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
func (queue *BoundedQueue[T]) Close() {
|
|
51
|
+
select {
|
|
52
|
+
case <-queue.closed:
|
|
53
|
+
default:
|
|
54
|
+
close(queue.closed)
|
|
55
|
+
}
|
|
56
|
+
}
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
package parallel
|
|
2
|
+
|
|
3
|
+
import (
|
|
4
|
+
"runtime"
|
|
5
|
+
"sync"
|
|
6
|
+
)
|
|
7
|
+
|
|
8
|
+
func ChunkCount(items int) int {
|
|
9
|
+
if items <= 0 {
|
|
10
|
+
return 0
|
|
11
|
+
}
|
|
12
|
+
workers := runtime.GOMAXPROCS(0)
|
|
13
|
+
if workers < 2 {
|
|
14
|
+
workers = 2
|
|
15
|
+
}
|
|
16
|
+
if workers > items {
|
|
17
|
+
workers = items
|
|
18
|
+
}
|
|
19
|
+
return workers
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
// ForEachChunk preserves result order by giving each callback a stable index.
|
|
23
|
+
// The caller owns result storage; callbacks must only write their own index.
|
|
24
|
+
func ForEachChunk(items int, callback func(index, start, end int)) {
|
|
25
|
+
workers := ChunkCount(items)
|
|
26
|
+
if workers == 0 {
|
|
27
|
+
return
|
|
28
|
+
}
|
|
29
|
+
chunkSize := (items + workers - 1) / workers
|
|
30
|
+
var wait sync.WaitGroup
|
|
31
|
+
for index := 0; index < workers; index++ {
|
|
32
|
+
start := index * chunkSize
|
|
33
|
+
end := start + chunkSize
|
|
34
|
+
if end > items {
|
|
35
|
+
end = items
|
|
36
|
+
}
|
|
37
|
+
if start >= end {
|
|
38
|
+
continue
|
|
39
|
+
}
|
|
40
|
+
wait.Add(1)
|
|
41
|
+
go func(index, start, end int) {
|
|
42
|
+
defer wait.Done()
|
|
43
|
+
callback(index, start, end)
|
|
44
|
+
}(index, start, end)
|
|
45
|
+
}
|
|
46
|
+
wait.Wait()
|
|
47
|
+
}
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
package parallel
|
|
2
|
+
|
|
3
|
+
import (
|
|
4
|
+
"context"
|
|
5
|
+
"sync"
|
|
6
|
+
)
|
|
7
|
+
|
|
8
|
+
type WorkerPool struct {
|
|
9
|
+
queue *BoundedQueue[func()]
|
|
10
|
+
wait sync.WaitGroup
|
|
11
|
+
closed chan struct{}
|
|
12
|
+
}
|
|
13
|
+
|
|
14
|
+
func NewWorkerPool(workers, queueSize int) *WorkerPool {
|
|
15
|
+
if workers < 1 {
|
|
16
|
+
workers = 1
|
|
17
|
+
}
|
|
18
|
+
pool := &WorkerPool{queue: NewBoundedQueue[func()](queueSize), closed: make(chan struct{})}
|
|
19
|
+
for index := 0; index < workers; index++ {
|
|
20
|
+
pool.wait.Add(1)
|
|
21
|
+
go func() {
|
|
22
|
+
defer pool.wait.Done()
|
|
23
|
+
for {
|
|
24
|
+
job, err := pool.queue.Pop(context.Background())
|
|
25
|
+
if err != nil {
|
|
26
|
+
return
|
|
27
|
+
}
|
|
28
|
+
job()
|
|
29
|
+
}
|
|
30
|
+
}()
|
|
31
|
+
}
|
|
32
|
+
return pool
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
func (pool *WorkerPool) Submit(ctx context.Context, job func()) error {
|
|
36
|
+
select {
|
|
37
|
+
case <-pool.closed:
|
|
38
|
+
return ErrQueueClosed
|
|
39
|
+
default:
|
|
40
|
+
}
|
|
41
|
+
return pool.queue.Push(ctx, job)
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
func (pool *WorkerPool) Close() {
|
|
45
|
+
select {
|
|
46
|
+
case <-pool.closed:
|
|
47
|
+
return
|
|
48
|
+
default:
|
|
49
|
+
close(pool.closed)
|
|
50
|
+
pool.queue.Close()
|
|
51
|
+
pool.wait.Wait()
|
|
52
|
+
}
|
|
53
|
+
}
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
package protocol
|
|
2
|
+
|
|
3
|
+
import (
|
|
4
|
+
"context"
|
|
5
|
+
"sync"
|
|
6
|
+
)
|
|
7
|
+
|
|
8
|
+
// CancellationRegistry gives the runtime a bounded, request-scoped context
|
|
9
|
+
// registry. A future multiplexed transport can cancel a running request by
|
|
10
|
+
// ID without changing the physical execution APIs.
|
|
11
|
+
type CancellationRegistry struct {
|
|
12
|
+
mu sync.Mutex
|
|
13
|
+
entries map[string]context.CancelFunc
|
|
14
|
+
}
|
|
15
|
+
|
|
16
|
+
func NewCancellationRegistry() *CancellationRegistry {
|
|
17
|
+
return &CancellationRegistry{entries: make(map[string]context.CancelFunc)}
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
func (registry *CancellationRegistry) Begin(parent context.Context, id string) context.Context {
|
|
21
|
+
ctx, cancel := context.WithCancel(parent)
|
|
22
|
+
registry.mu.Lock()
|
|
23
|
+
if previous, exists := registry.entries[id]; exists {
|
|
24
|
+
previous()
|
|
25
|
+
}
|
|
26
|
+
registry.entries[id] = cancel
|
|
27
|
+
registry.mu.Unlock()
|
|
28
|
+
return ctx
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
func (registry *CancellationRegistry) Cancel(id string) bool {
|
|
32
|
+
registry.mu.Lock()
|
|
33
|
+
cancel, exists := registry.entries[id]
|
|
34
|
+
if exists {
|
|
35
|
+
delete(registry.entries, id)
|
|
36
|
+
}
|
|
37
|
+
registry.mu.Unlock()
|
|
38
|
+
if exists {
|
|
39
|
+
cancel()
|
|
40
|
+
}
|
|
41
|
+
return exists
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
func (registry *CancellationRegistry) Finish(id string) {
|
|
45
|
+
registry.mu.Lock()
|
|
46
|
+
delete(registry.entries, id)
|
|
47
|
+
registry.mu.Unlock()
|
|
48
|
+
}
|
|
@@ -0,0 +1,263 @@
|
|
|
1
|
+
package protocol
|
|
2
|
+
|
|
3
|
+
import (
|
|
4
|
+
"bytes"
|
|
5
|
+
"encoding/binary"
|
|
6
|
+
"encoding/json"
|
|
7
|
+
"errors"
|
|
8
|
+
"math"
|
|
9
|
+
)
|
|
10
|
+
|
|
11
|
+
func EncodeRows(rows []map[string]interface{}) ([]byte, error) {
|
|
12
|
+
columns := make([]string, 0)
|
|
13
|
+
known := make(map[string]bool)
|
|
14
|
+
for _, row := range rows {
|
|
15
|
+
for key := range row {
|
|
16
|
+
if !known[key] {
|
|
17
|
+
known[key] = true
|
|
18
|
+
columns = append(columns, key)
|
|
19
|
+
}
|
|
20
|
+
}
|
|
21
|
+
}
|
|
22
|
+
if len(columns) > math.MaxUint16 || uint64(len(rows)) > math.MaxUint32 {
|
|
23
|
+
return nil, errors.New("binary rows batch is too large")
|
|
24
|
+
}
|
|
25
|
+
output := bytes.NewBuffer(make([]byte, 0, 6+len(columns)*16))
|
|
26
|
+
writeUint16(output, uint16(len(columns)))
|
|
27
|
+
writeUint32(output, uint32(len(rows)))
|
|
28
|
+
for _, column := range columns {
|
|
29
|
+
if len(column) > math.MaxUint16 {
|
|
30
|
+
return nil, errors.New("binary row column name is too long")
|
|
31
|
+
}
|
|
32
|
+
writeUint16(output, uint16(len(column)))
|
|
33
|
+
output.WriteString(column)
|
|
34
|
+
}
|
|
35
|
+
for _, column := range columns {
|
|
36
|
+
for _, row := range rows {
|
|
37
|
+
if err := encodeValue(output, row[column]); err != nil {
|
|
38
|
+
return nil, err
|
|
39
|
+
}
|
|
40
|
+
}
|
|
41
|
+
}
|
|
42
|
+
return output.Bytes(), nil
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
func DecodeRows(data []byte) ([]map[string]interface{}, error) {
|
|
46
|
+
reader := newValueReader(data)
|
|
47
|
+
columnCount, err := reader.uint16()
|
|
48
|
+
if err != nil {
|
|
49
|
+
return nil, errors.New("binary rows batch is truncated")
|
|
50
|
+
}
|
|
51
|
+
rowCount, err := reader.uint32()
|
|
52
|
+
if err != nil {
|
|
53
|
+
return nil, errors.New("binary rows batch is truncated")
|
|
54
|
+
}
|
|
55
|
+
columns := make([]string, columnCount)
|
|
56
|
+
for index := range columns {
|
|
57
|
+
name, err := reader.string16()
|
|
58
|
+
if err != nil {
|
|
59
|
+
return nil, errors.New("binary row column metadata is invalid")
|
|
60
|
+
}
|
|
61
|
+
columns[index] = name
|
|
62
|
+
}
|
|
63
|
+
rows := make([]map[string]interface{}, rowCount)
|
|
64
|
+
for index := range rows {
|
|
65
|
+
rows[index] = make(map[string]interface{}, columnCount)
|
|
66
|
+
}
|
|
67
|
+
for _, column := range columns {
|
|
68
|
+
for rowIndex := range rows {
|
|
69
|
+
value, err := reader.value()
|
|
70
|
+
if err != nil {
|
|
71
|
+
return nil, errors.New("binary row value is invalid")
|
|
72
|
+
}
|
|
73
|
+
rows[rowIndex][column] = value
|
|
74
|
+
}
|
|
75
|
+
}
|
|
76
|
+
if reader.remaining() != 0 {
|
|
77
|
+
return nil, errors.New("binary rows batch has trailing bytes")
|
|
78
|
+
}
|
|
79
|
+
return rows, nil
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
func encodeValue(output *bytes.Buffer, value interface{}) error {
|
|
83
|
+
switch typed := value.(type) {
|
|
84
|
+
case nil:
|
|
85
|
+
output.WriteByte(0)
|
|
86
|
+
case bool:
|
|
87
|
+
output.WriteByte(1)
|
|
88
|
+
if typed {
|
|
89
|
+
output.WriteByte(1)
|
|
90
|
+
} else {
|
|
91
|
+
output.WriteByte(0)
|
|
92
|
+
}
|
|
93
|
+
case json.Number:
|
|
94
|
+
if integer, err := typed.Int64(); err == nil {
|
|
95
|
+
output.WriteByte(2)
|
|
96
|
+
writeInt64(output, integer)
|
|
97
|
+
} else if number, err := typed.Float64(); err == nil {
|
|
98
|
+
output.WriteByte(3)
|
|
99
|
+
writeFloat64(output, number)
|
|
100
|
+
} else {
|
|
101
|
+
return errors.New("invalid numeric value")
|
|
102
|
+
}
|
|
103
|
+
case int:
|
|
104
|
+
output.WriteByte(2)
|
|
105
|
+
writeInt64(output, int64(typed))
|
|
106
|
+
case int8:
|
|
107
|
+
output.WriteByte(2)
|
|
108
|
+
writeInt64(output, int64(typed))
|
|
109
|
+
case int16:
|
|
110
|
+
output.WriteByte(2)
|
|
111
|
+
writeInt64(output, int64(typed))
|
|
112
|
+
case int32:
|
|
113
|
+
output.WriteByte(2)
|
|
114
|
+
writeInt64(output, int64(typed))
|
|
115
|
+
case int64:
|
|
116
|
+
output.WriteByte(2)
|
|
117
|
+
writeInt64(output, typed)
|
|
118
|
+
case uint:
|
|
119
|
+
output.WriteByte(2)
|
|
120
|
+
writeInt64(output, int64(typed))
|
|
121
|
+
case uint8:
|
|
122
|
+
output.WriteByte(2)
|
|
123
|
+
writeInt64(output, int64(typed))
|
|
124
|
+
case uint16:
|
|
125
|
+
output.WriteByte(2)
|
|
126
|
+
writeInt64(output, int64(typed))
|
|
127
|
+
case uint32:
|
|
128
|
+
output.WriteByte(2)
|
|
129
|
+
writeInt64(output, int64(typed))
|
|
130
|
+
case uint64:
|
|
131
|
+
if typed > math.MaxInt64 {
|
|
132
|
+
return errors.New("unsigned numeric value is too large")
|
|
133
|
+
}
|
|
134
|
+
output.WriteByte(2)
|
|
135
|
+
writeInt64(output, int64(typed))
|
|
136
|
+
case float32:
|
|
137
|
+
output.WriteByte(3)
|
|
138
|
+
writeFloat64(output, float64(typed))
|
|
139
|
+
case float64:
|
|
140
|
+
output.WriteByte(3)
|
|
141
|
+
writeFloat64(output, typed)
|
|
142
|
+
case []byte:
|
|
143
|
+
output.WriteByte(5)
|
|
144
|
+
writeBytes(output, typed)
|
|
145
|
+
case string:
|
|
146
|
+
output.WriteByte(4)
|
|
147
|
+
writeBytes(output, []byte(typed))
|
|
148
|
+
default:
|
|
149
|
+
encoded, err := json.Marshal(value)
|
|
150
|
+
if err != nil {
|
|
151
|
+
return errors.New("value is not serializable")
|
|
152
|
+
}
|
|
153
|
+
output.WriteByte(6)
|
|
154
|
+
writeBytes(output, encoded)
|
|
155
|
+
}
|
|
156
|
+
return nil
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
type valueReader struct {
|
|
160
|
+
data []byte
|
|
161
|
+
offset int
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
func newValueReader(data []byte) *valueReader { return &valueReader{data: data} }
|
|
165
|
+
func (reader *valueReader) remaining() int { return len(reader.data) - reader.offset }
|
|
166
|
+
|
|
167
|
+
func (reader *valueReader) take(length int) ([]byte, error) {
|
|
168
|
+
if length < 0 || reader.remaining() < length {
|
|
169
|
+
return nil, errors.New("binary value is truncated")
|
|
170
|
+
}
|
|
171
|
+
value := reader.data[reader.offset : reader.offset+length]
|
|
172
|
+
reader.offset += length
|
|
173
|
+
return value, nil
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
func (reader *valueReader) uint16() (uint16, error) {
|
|
177
|
+
value, err := reader.take(2)
|
|
178
|
+
if err != nil {
|
|
179
|
+
return 0, err
|
|
180
|
+
}
|
|
181
|
+
return binary.LittleEndian.Uint16(value), nil
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
func (reader *valueReader) uint32() (uint32, error) {
|
|
185
|
+
value, err := reader.take(4)
|
|
186
|
+
if err != nil {
|
|
187
|
+
return 0, err
|
|
188
|
+
}
|
|
189
|
+
return binary.LittleEndian.Uint32(value), nil
|
|
190
|
+
}
|
|
191
|
+
|
|
192
|
+
func (reader *valueReader) string16() (string, error) {
|
|
193
|
+
length, err := reader.uint16()
|
|
194
|
+
if err != nil {
|
|
195
|
+
return "", err
|
|
196
|
+
}
|
|
197
|
+
value, err := reader.take(int(length))
|
|
198
|
+
return string(value), err
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
func (reader *valueReader) value() (interface{}, error) {
|
|
202
|
+
tag, err := reader.take(1)
|
|
203
|
+
if err != nil {
|
|
204
|
+
return nil, err
|
|
205
|
+
}
|
|
206
|
+
switch tag[0] {
|
|
207
|
+
case 0:
|
|
208
|
+
return nil, nil
|
|
209
|
+
case 1:
|
|
210
|
+
value, err := reader.take(1)
|
|
211
|
+
return value[0] == 1, err
|
|
212
|
+
case 2:
|
|
213
|
+
value, err := reader.take(8)
|
|
214
|
+
if err != nil {
|
|
215
|
+
return nil, err
|
|
216
|
+
}
|
|
217
|
+
return int64(binary.LittleEndian.Uint64(value)), nil
|
|
218
|
+
case 3:
|
|
219
|
+
value, err := reader.take(8)
|
|
220
|
+
if err != nil {
|
|
221
|
+
return nil, err
|
|
222
|
+
}
|
|
223
|
+
return math.Float64frombits(binary.LittleEndian.Uint64(value)), nil
|
|
224
|
+
case 4, 5, 6:
|
|
225
|
+
length, err := reader.uint32()
|
|
226
|
+
if err != nil {
|
|
227
|
+
return nil, err
|
|
228
|
+
}
|
|
229
|
+
value, err := reader.take(int(length))
|
|
230
|
+
if err != nil {
|
|
231
|
+
return nil, err
|
|
232
|
+
}
|
|
233
|
+
if tag[0] == 4 {
|
|
234
|
+
return string(value), nil
|
|
235
|
+
}
|
|
236
|
+
if tag[0] == 5 {
|
|
237
|
+
return append([]byte(nil), value...), nil
|
|
238
|
+
}
|
|
239
|
+
var decoded interface{}
|
|
240
|
+
if err := json.Unmarshal(value, &decoded); err != nil {
|
|
241
|
+
return nil, err
|
|
242
|
+
}
|
|
243
|
+
return decoded, nil
|
|
244
|
+
default:
|
|
245
|
+
return nil, errors.New("unknown binary value tag")
|
|
246
|
+
}
|
|
247
|
+
}
|
|
248
|
+
|
|
249
|
+
func writeUint16(output *bytes.Buffer, value uint16) {
|
|
250
|
+
_ = binary.Write(output, binary.LittleEndian, value)
|
|
251
|
+
}
|
|
252
|
+
func writeUint32(output *bytes.Buffer, value uint32) {
|
|
253
|
+
_ = binary.Write(output, binary.LittleEndian, value)
|
|
254
|
+
}
|
|
255
|
+
func writeBytes(output *bytes.Buffer, value []byte) {
|
|
256
|
+
writeUint32(output, uint32(len(value)))
|
|
257
|
+
output.Write(value)
|
|
258
|
+
}
|
|
259
|
+
func writeInt64(output *bytes.Buffer, value int64) { writeUint64(output, uint64(value)) }
|
|
260
|
+
func writeFloat64(output *bytes.Buffer, value float64) { writeUint64(output, math.Float64bits(value)) }
|
|
261
|
+
func writeUint64(output *bytes.Buffer, value uint64) {
|
|
262
|
+
_ = binary.Write(output, binary.LittleEndian, value)
|
|
263
|
+
}
|