logstash-output-cassandra-v5 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +7 -0
- data/CONTRIBUTORS +10 -0
- data/Gemfile +4 -0
- data/LICENSE +13 -0
- data/README.md +146 -0
- data/lib/logstash/outputs/cassandra/backoff_retry_policy.rb +65 -0
- data/lib/logstash/outputs/cassandra/buffer.rb +125 -0
- data/lib/logstash/outputs/cassandra/event_parser.rb +164 -0
- data/lib/logstash/outputs/cassandra/safe_submitter.rb +121 -0
- data/lib/logstash/outputs/cassandra/schema_fetcher_version_patch.rb +23 -0
- data/lib/logstash/outputs/cassandra.rb +166 -0
- data/logstash-output-cassandra-v5.gemspec +32 -0
- data/spec/cassandra_spec_helper.rb +14 -0
- data/spec/integration/outputs/cassandra_spec.rb +114 -0
- data/spec/integration/outputs/integration_helper.rb +91 -0
- data/spec/unit/outputs/backoff_retry_policy_spec.rb +131 -0
- data/spec/unit/outputs/buffer_spec.rb +119 -0
- data/spec/unit/outputs/cassandra_spec.rb +5 -0
- data/spec/unit/outputs/event_parser_spec.rb +311 -0
- data/spec/unit/outputs/safe_submitter_spec.rb +210 -0
- metadata +222 -0
checksums.yaml
ADDED
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
---
|
|
2
|
+
SHA256:
|
|
3
|
+
metadata.gz: 5a1152690c197d94b35534062a0fd5d017db79e699a0bc30cd97b39f04ddf8e3
|
|
4
|
+
data.tar.gz: 8ca6fc3784f815ff823df65ef73e9e8b3f99df3b2eacd50e31b3c08536a8f3fd
|
|
5
|
+
SHA512:
|
|
6
|
+
metadata.gz: 9e83ee3db069423a55c37d76bcd121454f979e08ba15de74907662836d57e2ed9197c36807334eacfe2279e326271b1344e72d8f45390c8c5d7379b680322d47
|
|
7
|
+
data.tar.gz: 1e7549cb53ab649f6c2d0f01c35f52bd61914e708dc59e41eaf3165730ac9a775978e7f4460a296f08373eba27522cf5069aed939acdf600fbb1c3b84401240b
|
data/CONTRIBUTORS
ADDED
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
The following is a list of people who have contributed (in chronological order) ideas, code, bug
|
|
2
|
+
reports, or in general have helped this plugin along its way.
|
|
3
|
+
|
|
4
|
+
Contributors:
|
|
5
|
+
* Luis Burbano Ulloa (lburbanoulloa)
|
|
6
|
+
* Oleg Tokarev (otokarev)
|
|
7
|
+
* Elad Amit (eladamitpxi, amitelad7)
|
|
8
|
+
* Valentin Fischer (valentinul)
|
|
9
|
+
* tansinee
|
|
10
|
+
* Pitsanu Swangpheaw (roongr2k7)
|
data/Gemfile
ADDED
data/LICENSE
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
Copyright (c) 2016 PerimeterX <http://www.perimeterx.com>
|
|
2
|
+
|
|
3
|
+
Licensed under the Apache License, Version 2.0 (the "License");
|
|
4
|
+
you may not use this file except in compliance with the License.
|
|
5
|
+
You may obtain a copy of the License at
|
|
6
|
+
|
|
7
|
+
http://www.apache.org/licenses/LICENSE-2.0
|
|
8
|
+
|
|
9
|
+
Unless required by applicable law or agreed to in writing, software
|
|
10
|
+
distributed under the License is distributed on an "AS IS" BASIS,
|
|
11
|
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
12
|
+
See the License for the specific language governing permissions and
|
|
13
|
+
limitations under the License.
|
data/README.md
ADDED
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
# Logstash Cassandra Output Plugin v5
|
|
2
|
+
|
|
3
|
+
This is a plugin for [Logstash](https://github.com/elastic/logstash).
|
|
4
|
+
|
|
5
|
+
It is fully free and fully open source. The license is Apache 2.0, meaning you are pretty much free to use it however you want in whatever way.
|
|
6
|
+
|
|
7
|
+
It was originally a fork of the [logstash-output-cassandra](https://github.com/PerimeterX/logstash-output-cassandra
|
|
8
|
+
|
|
9
|
+
This version update the plugin to work with Cassandra v5, and fix an error when inserting UUId fields.
|
|
10
|
+
|
|
11
|
+
## Usage
|
|
12
|
+
|
|
13
|
+
<pre><code>
|
|
14
|
+
output {
|
|
15
|
+
cassandra {
|
|
16
|
+
# List of Cassandra hostname(s) or IP-address(es)
|
|
17
|
+
hosts => [ "cass-01", "cass-02" ]
|
|
18
|
+
|
|
19
|
+
# The port cassandra is listening to
|
|
20
|
+
port => 9042
|
|
21
|
+
|
|
22
|
+
# The protocol version to use with cassandra
|
|
23
|
+
protocol_version => 4
|
|
24
|
+
|
|
25
|
+
# Cassandra consistency level.
|
|
26
|
+
# Options: "any", "one", "two", "three", "quorum", "all", "local_quorum", "each_quorum", "serial", "local_serial", "local_one"
|
|
27
|
+
# Default: "one"
|
|
28
|
+
consistency => 'any'
|
|
29
|
+
|
|
30
|
+
# The keyspace to use
|
|
31
|
+
keyspace => "a_ks"
|
|
32
|
+
|
|
33
|
+
# The table to use (event level processing (e.g. %{[key]}) is supported)
|
|
34
|
+
table => "%{[@metadata][cassandra_table]}"
|
|
35
|
+
|
|
36
|
+
# Username
|
|
37
|
+
username => "cassandra"
|
|
38
|
+
|
|
39
|
+
# Password
|
|
40
|
+
password => "cassandra"
|
|
41
|
+
|
|
42
|
+
# An optional hints hash which will be used in case filter_transform or filter_transform_event_key are not in use
|
|
43
|
+
# It is used to trigger a forced type casting to the cassandra driver types in
|
|
44
|
+
# the form of a hash from column name to type name in the following manner:
|
|
45
|
+
hints => {
|
|
46
|
+
id => "int"
|
|
47
|
+
at => "timestamp"
|
|
48
|
+
resellerId => "int"
|
|
49
|
+
errno => "int"
|
|
50
|
+
duration => "float"
|
|
51
|
+
ip => "inet"
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
# The retry policy to use (the default is the default retry policy)
|
|
55
|
+
# the hash requires the name of the policy and the params it requires
|
|
56
|
+
# The available policy names are:
|
|
57
|
+
# * default => retry once if needed / possible
|
|
58
|
+
# * downgrading_consistency => retry once with a best guess lowered consistency
|
|
59
|
+
# * failthrough => fail immediately (i.e. no retries)
|
|
60
|
+
# * backoff => a version of the default retry policy but with configurable backoff retries
|
|
61
|
+
# The backoff options are as follows:
|
|
62
|
+
# * backoff_type => either * or ** for linear and exponential backoffs respectively
|
|
63
|
+
# * backoff_size => the left operand for the backoff type in seconds
|
|
64
|
+
# * retry_limit => the maximum amount of retries to allow per query
|
|
65
|
+
# example:
|
|
66
|
+
# using { "type" => "backoff" "backoff_type" => "**" "backoff_size" => 2 "retry_limit" => 10 } will perform 10 retries with the following wait times: 1, 2, 4, 8, 16, ... 1024
|
|
67
|
+
# NOTE: there is an underlying assumption that the insert query is idempotent !!!
|
|
68
|
+
# NOTE: when the backoff retry policy is used, it will also be used to handle pure client timeouts and not just ones coming from the coordinator
|
|
69
|
+
retry_policy => { "type" => "default" }
|
|
70
|
+
|
|
71
|
+
# The command execution timeout
|
|
72
|
+
request_timeout => 1
|
|
73
|
+
|
|
74
|
+
# Ignore bad values
|
|
75
|
+
ignore_bad_values => false
|
|
76
|
+
|
|
77
|
+
# In Logstashes >= 2.2 this setting defines the maximum sized bulk request Logstash will make
|
|
78
|
+
# You you may want to increase this to be in line with your pipeline's batch size.
|
|
79
|
+
# If you specify a number larger than the batch size of your pipeline it will have no effect,
|
|
80
|
+
# save for the case where a filter increases the size of an inflight batch by outputting
|
|
81
|
+
# events.
|
|
82
|
+
#
|
|
83
|
+
# In Logstashes <= 2.1 this plugin uses its own internal buffer of events.
|
|
84
|
+
# This config option sets that size. In these older logstashes this size may
|
|
85
|
+
# have a significant impact on heap usage, whereas in 2.2+ it will never increase it.
|
|
86
|
+
# To make efficient bulk API calls, we will buffer a certain number of
|
|
87
|
+
# events before flushing that out to Cassandra. This setting
|
|
88
|
+
# controls how many events will be buffered before sending a batch
|
|
89
|
+
# of events. Increasing the `flush_size` has an effect on Logstash's heap size.
|
|
90
|
+
# Remember to also increase the heap size using `LS_HEAP_SIZE` if you are sending big commands
|
|
91
|
+
# or have increased the `flush_size` to a higher value.
|
|
92
|
+
flush_size => 500
|
|
93
|
+
|
|
94
|
+
# The amount of time since last flush before a flush is forced.
|
|
95
|
+
#
|
|
96
|
+
# This setting helps ensure slow event rates don't get stuck in Logstash.
|
|
97
|
+
# For example, if your `flush_size` is 100, and you have received 10 events,
|
|
98
|
+
# and it has been more than `idle_flush_time` seconds since the last flush,
|
|
99
|
+
# Logstash will flush those 10 events automatically.
|
|
100
|
+
#
|
|
101
|
+
# This helps keep both fast and slow log streams moving along in
|
|
102
|
+
# near-real-time.
|
|
103
|
+
idle_flush_time => 1
|
|
104
|
+
}
|
|
105
|
+
}
|
|
106
|
+
</code></pre>
|
|
107
|
+
|
|
108
|
+
## Running Plugin in Logstash
|
|
109
|
+
### Run in a local Logstash clone
|
|
110
|
+
|
|
111
|
+
Edit Logstash Gemfile and add the local plugin path, for example:
|
|
112
|
+
```
|
|
113
|
+
gem "logstash-output-cassandra", :path => "/your/local/logstash-output-cassandra"
|
|
114
|
+
```
|
|
115
|
+
And install by executing:
|
|
116
|
+
```
|
|
117
|
+
bin/plugin install --no-verify
|
|
118
|
+
```
|
|
119
|
+
|
|
120
|
+
Or install plugin from RubyGems:
|
|
121
|
+
```
|
|
122
|
+
bin/plugin install logstash-output-cassandra
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
And then run Logstash with the plugin:
|
|
126
|
+
```
|
|
127
|
+
bin/logstash -e 'output {cassandra {}}'
|
|
128
|
+
```
|
|
129
|
+
|
|
130
|
+
### Run in an installed Logstash
|
|
131
|
+
|
|
132
|
+
You can use the same method to run your plugin in an installed Logstash by editing its Gemfile and pointing the :path to your local plugin development directory or you can build the gem and install it using:
|
|
133
|
+
|
|
134
|
+
Build your plugin gem
|
|
135
|
+
```
|
|
136
|
+
gem build logstash-output-cassandra.gemspec
|
|
137
|
+
```
|
|
138
|
+
Install the plugin from the Logstash home
|
|
139
|
+
```
|
|
140
|
+
bin/plugin install /your/local/plugin/logstash-output-cassandra.gem
|
|
141
|
+
```
|
|
142
|
+
Run Logstash with the plugin
|
|
143
|
+
```
|
|
144
|
+
bin/logstash -e 'output {cassandra {}}'
|
|
145
|
+
```
|
|
146
|
+
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
# encoding: utf-8
|
|
2
|
+
require 'cassandra'
|
|
3
|
+
|
|
4
|
+
module Cassandra
|
|
5
|
+
module Retry
|
|
6
|
+
module Policies
|
|
7
|
+
# This is a version of the default retry policy (https://github.com/datastax/ruby-driver/blob/v2.1.5/lib/cassandra/retry/policies/default.rb)
|
|
8
|
+
# with backoff retry configuration options
|
|
9
|
+
class Backoff
|
|
10
|
+
include ::Cassandra::Retry::Policy
|
|
11
|
+
|
|
12
|
+
def initialize(opts)
|
|
13
|
+
@logger = opts['logger']
|
|
14
|
+
@backoff_type = opts['backoff_type']
|
|
15
|
+
@backoff_size = opts['backoff_size']
|
|
16
|
+
@retry_limit = opts['retry_limit']
|
|
17
|
+
end
|
|
18
|
+
|
|
19
|
+
def read_timeout(statement, consistency, required, received, retrieved, retries)
|
|
20
|
+
retry_with_backoff({ :statement => statement, :consistency => consistency, :required => required,
|
|
21
|
+
:received => received, :retrieved => retrieved, :retries => retries })
|
|
22
|
+
end
|
|
23
|
+
|
|
24
|
+
def write_timeout(statement, consistency, type, required, received, retries)
|
|
25
|
+
retry_with_backoff({ :statement => statement, :consistency => consistency, :type => type,
|
|
26
|
+
:required => required, :received => received, :retries => retries })
|
|
27
|
+
end
|
|
28
|
+
|
|
29
|
+
def unavailable(statement, consistency, required, alive, retries)
|
|
30
|
+
retry_with_backoff({ :statement => statement, :consistency => consistency, :required => required,
|
|
31
|
+
:alive => alive, :retries => retries })
|
|
32
|
+
end
|
|
33
|
+
|
|
34
|
+
def retry_with_backoff(opts)
|
|
35
|
+
if @retry_limit > -1 && opts[:retries] > @retry_limit
|
|
36
|
+
@logger.error('backoff retries exhausted', :opts => opts)
|
|
37
|
+
return reraise
|
|
38
|
+
end
|
|
39
|
+
|
|
40
|
+
@logger.error('activating backoff wait', :opts => opts)
|
|
41
|
+
backoff_wait_before_next_retry(opts[:retries])
|
|
42
|
+
|
|
43
|
+
try_again(opts[:consistency])
|
|
44
|
+
end
|
|
45
|
+
|
|
46
|
+
private
|
|
47
|
+
def backoff_wait_before_next_retry(retries)
|
|
48
|
+
backoff_wait_time = calculate_backoff_wait_time(retries)
|
|
49
|
+
Kernel::sleep(backoff_wait_time)
|
|
50
|
+
end
|
|
51
|
+
|
|
52
|
+
def calculate_backoff_wait_time(retries)
|
|
53
|
+
case @backoff_type
|
|
54
|
+
when '**'
|
|
55
|
+
return @backoff_size ** retries
|
|
56
|
+
when '*'
|
|
57
|
+
return @backoff_size * retries
|
|
58
|
+
else
|
|
59
|
+
raise ArgumentError, "unknown backoff type #{@backoff_type}"
|
|
60
|
+
end
|
|
61
|
+
end
|
|
62
|
+
end
|
|
63
|
+
end
|
|
64
|
+
end
|
|
65
|
+
end
|
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
# encoding: utf-8
|
|
2
|
+
require "concurrent"
|
|
3
|
+
java_import java.util.concurrent.locks.ReentrantLock
|
|
4
|
+
|
|
5
|
+
module LogStash; module Outputs; module Cassandra
|
|
6
|
+
class Buffer
|
|
7
|
+
def initialize(logger, max_size, flush_interval, &block)
|
|
8
|
+
@logger = logger
|
|
9
|
+
# You need to aquire this for anything modifying state generally
|
|
10
|
+
@operations_mutex = Mutex.new
|
|
11
|
+
@operations_lock = java.util.concurrent.locks.ReentrantLock.new
|
|
12
|
+
|
|
13
|
+
@stopping = Concurrent::AtomicBoolean.new(false)
|
|
14
|
+
@max_size = max_size
|
|
15
|
+
@submit_proc = block
|
|
16
|
+
|
|
17
|
+
@buffer = []
|
|
18
|
+
|
|
19
|
+
@last_flush = Time.now
|
|
20
|
+
@flush_interval = flush_interval
|
|
21
|
+
@flush_thread = spawn_interval_flusher
|
|
22
|
+
end
|
|
23
|
+
|
|
24
|
+
def push(item)
|
|
25
|
+
synchronize do |buffer|
|
|
26
|
+
push_unsafe(item)
|
|
27
|
+
end
|
|
28
|
+
end
|
|
29
|
+
alias_method :<<, :push
|
|
30
|
+
|
|
31
|
+
# Push multiple items onto the buffer in a single operation
|
|
32
|
+
def push_multi(items)
|
|
33
|
+
raise ArgumentError, "push multi takes an array!, not an #{items.class}!" unless items.is_a?(Array)
|
|
34
|
+
synchronize do |buffer|
|
|
35
|
+
items.each {|item| push_unsafe(item) }
|
|
36
|
+
end
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
def flush
|
|
40
|
+
synchronize { flush_unsafe }
|
|
41
|
+
end
|
|
42
|
+
|
|
43
|
+
def stop(do_flush=true,wait_complete=true)
|
|
44
|
+
return if stopping?
|
|
45
|
+
@stopping.make_true
|
|
46
|
+
|
|
47
|
+
# No need to acquire a lock in this case
|
|
48
|
+
return if !do_flush && !wait_complete
|
|
49
|
+
|
|
50
|
+
synchronize do
|
|
51
|
+
flush_unsafe if do_flush
|
|
52
|
+
@flush_thread.join if wait_complete
|
|
53
|
+
end
|
|
54
|
+
end
|
|
55
|
+
|
|
56
|
+
def contents
|
|
57
|
+
synchronize {|buffer| buffer}
|
|
58
|
+
end
|
|
59
|
+
|
|
60
|
+
# For externally operating on the buffer contents
|
|
61
|
+
# this takes a block and will yield the internal buffer and executes
|
|
62
|
+
# the block in a synchronized block from the internal mutex
|
|
63
|
+
def synchronize
|
|
64
|
+
@operations_mutex.synchronize { yield(@buffer) }
|
|
65
|
+
end
|
|
66
|
+
|
|
67
|
+
# These methods are private for various reasons, chief among them threadsafety!
|
|
68
|
+
# Many require the @operations_mutex to be locked to be safe
|
|
69
|
+
private
|
|
70
|
+
|
|
71
|
+
def push_unsafe(item)
|
|
72
|
+
@buffer << item
|
|
73
|
+
if @buffer.size >= @max_size
|
|
74
|
+
flush_unsafe
|
|
75
|
+
end
|
|
76
|
+
end
|
|
77
|
+
|
|
78
|
+
def spawn_interval_flusher
|
|
79
|
+
Thread.new do
|
|
80
|
+
loop do
|
|
81
|
+
sleep 0.2
|
|
82
|
+
break if stopping?
|
|
83
|
+
synchronize { interval_flush }
|
|
84
|
+
end
|
|
85
|
+
end
|
|
86
|
+
end
|
|
87
|
+
|
|
88
|
+
def interval_flush
|
|
89
|
+
if last_flush_seconds_ago >= @flush_interval
|
|
90
|
+
begin
|
|
91
|
+
@logger.debug? && @logger.debug("Flushing buffer at interval",
|
|
92
|
+
:instance => self.inspect,
|
|
93
|
+
:interval => @flush_interval)
|
|
94
|
+
flush_unsafe
|
|
95
|
+
rescue StandardError => e
|
|
96
|
+
@logger.warn("Error flushing buffer at interval!",
|
|
97
|
+
:instance => self.inspect,
|
|
98
|
+
:message => e.message,
|
|
99
|
+
:class => e.class.name,
|
|
100
|
+
:backtrace => e.backtrace
|
|
101
|
+
)
|
|
102
|
+
rescue Exception => e
|
|
103
|
+
@logger.warn("Exception flushing buffer at interval!", :error => e.message, :class => e.class.name)
|
|
104
|
+
end
|
|
105
|
+
end
|
|
106
|
+
end
|
|
107
|
+
|
|
108
|
+
def flush_unsafe
|
|
109
|
+
if @buffer.size > 0
|
|
110
|
+
@submit_proc.call(@buffer)
|
|
111
|
+
@buffer.clear
|
|
112
|
+
end
|
|
113
|
+
|
|
114
|
+
@last_flush = Time.now # This must always be set to ensure correct timer behavior
|
|
115
|
+
end
|
|
116
|
+
|
|
117
|
+
def last_flush_seconds_ago
|
|
118
|
+
Time.now - @last_flush
|
|
119
|
+
end
|
|
120
|
+
|
|
121
|
+
def stopping?
|
|
122
|
+
@stopping.true?
|
|
123
|
+
end
|
|
124
|
+
end
|
|
125
|
+
end end end
|
|
@@ -0,0 +1,164 @@
|
|
|
1
|
+
# encoding: utf-8
|
|
2
|
+
require 'time'
|
|
3
|
+
require 'cassandra'
|
|
4
|
+
|
|
5
|
+
module LogStash; module Outputs; module Cassandra
|
|
6
|
+
# Responsible for accepting events from the pipeline and returning actions for the SafeSubmitter
|
|
7
|
+
class EventParser
|
|
8
|
+
def initialize(options)
|
|
9
|
+
@logger = options['logger']
|
|
10
|
+
@table = options['table']
|
|
11
|
+
@filter_transform_event_key = options['filter_transform_event_key']
|
|
12
|
+
assert_filter_transform_structure(options['filter_transform']) if options['filter_transform']
|
|
13
|
+
@filter_transform = options['filter_transform']
|
|
14
|
+
@hints = options['hints']
|
|
15
|
+
@ignore_bad_values = options['ignore_bad_values']
|
|
16
|
+
end
|
|
17
|
+
|
|
18
|
+
def parse(event)
|
|
19
|
+
action = {}
|
|
20
|
+
begin
|
|
21
|
+
action['table'] = event.sprintf(@table)
|
|
22
|
+
filter_transform = get_filter_transform(event)
|
|
23
|
+
if filter_transform
|
|
24
|
+
action['data'] = {}
|
|
25
|
+
filter_transform.each { |filter|
|
|
26
|
+
add_event_value_from_filter_to_action(event, filter, action)
|
|
27
|
+
}
|
|
28
|
+
else
|
|
29
|
+
add_event_data_using_configured_hints(event, action)
|
|
30
|
+
end
|
|
31
|
+
@logger.debug('event parsed to action', :action => action)
|
|
32
|
+
rescue Exception => e
|
|
33
|
+
@logger.error('failed parsing event', :event => event, :error => e)
|
|
34
|
+
action = nil
|
|
35
|
+
end
|
|
36
|
+
action
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
private
|
|
40
|
+
def get_filter_transform(event)
|
|
41
|
+
filter_transform = nil
|
|
42
|
+
if @filter_transform_event_key
|
|
43
|
+
filter_transform = event.get(@filter_transform_event_key)
|
|
44
|
+
assert_filter_transform_structure(filter_transform)
|
|
45
|
+
elsif @filter_transform.length > 0
|
|
46
|
+
filter_transform = @filter_transform
|
|
47
|
+
end
|
|
48
|
+
filter_transform
|
|
49
|
+
end
|
|
50
|
+
|
|
51
|
+
def assert_filter_transform_structure(filter_transform)
|
|
52
|
+
filter_transform.each { |item|
|
|
53
|
+
if !item.has_key?('event_key') || !item.has_key?('column_name')
|
|
54
|
+
raise ArgumentError, "item is incorrectly configured in filter_transform:\nitem => #{item}\nfilter_transform => #{filter_transform}"
|
|
55
|
+
end
|
|
56
|
+
}
|
|
57
|
+
end
|
|
58
|
+
|
|
59
|
+
def add_event_value_from_filter_to_action(event, filter, action)
|
|
60
|
+
event_data = event.sprintf(filter['event_key'])
|
|
61
|
+
unless filter.fetch('expansion_only', false)
|
|
62
|
+
event_data = event.get(event_data)
|
|
63
|
+
end
|
|
64
|
+
if filter.has_key?('cassandra_type')
|
|
65
|
+
cassandra_type = event.sprintf(filter['cassandra_type'])
|
|
66
|
+
event_data = convert_value_to_cassandra_type_or_default_if_configured(event_data, cassandra_type)
|
|
67
|
+
end
|
|
68
|
+
column_name = event.sprintf(filter['column_name'])
|
|
69
|
+
action['data'][column_name] = event_data
|
|
70
|
+
end
|
|
71
|
+
|
|
72
|
+
def add_event_data_using_configured_hints(event, action)
|
|
73
|
+
action_data = event.to_hash.reject { |key| %r{^@} =~ key }
|
|
74
|
+
|
|
75
|
+
@hints.each do |event_key, cassandra_type|
|
|
76
|
+
if action_data.has_key?(event_key)
|
|
77
|
+
action_data[event_key] = convert_value_to_cassandra_type_or_default_if_configured(action_data[event_key], cassandra_type)
|
|
78
|
+
end
|
|
79
|
+
end
|
|
80
|
+
action['data'] = action_data
|
|
81
|
+
end
|
|
82
|
+
|
|
83
|
+
def convert_value_to_cassandra_type_or_default_if_configured(event_data, cassandra_type)
|
|
84
|
+
typed_event_data = nil
|
|
85
|
+
begin
|
|
86
|
+
typed_event_data = convert_value_to_cassandra_type(event_data, cassandra_type)
|
|
87
|
+
rescue Exception => e
|
|
88
|
+
error_message = "Cannot convert `value (`#{event_data}`) to `#{cassandra_type}` type"
|
|
89
|
+
if @ignore_bad_values
|
|
90
|
+
case cassandra_type
|
|
91
|
+
when 'float', 'int', 'varint', 'bigint', 'double', 'counter', 'timestamp'
|
|
92
|
+
typed_event_data = convert_value_to_cassandra_type(0, cassandra_type)
|
|
93
|
+
when 'uuid', 'timeuuid'
|
|
94
|
+
typed_event_data = convert_value_to_cassandra_type('00000000-0000-0000-0000-000000000000', cassandra_type)
|
|
95
|
+
when 'inet'
|
|
96
|
+
typed_event_data = convert_value_to_cassandra_type('0.0.0.0', cassandra_type)
|
|
97
|
+
when /^set<.*>$/
|
|
98
|
+
typed_event_data = convert_value_to_cassandra_type([], cassandra_type)
|
|
99
|
+
else
|
|
100
|
+
raise ArgumentError, "unable to provide a default value for type #{event_data}"
|
|
101
|
+
end
|
|
102
|
+
@logger.warn(error_message, :exception => e, :backtrace => e.backtrace)
|
|
103
|
+
else
|
|
104
|
+
@logger.error(error_message, :exception => e, :backtrace => e.backtrace)
|
|
105
|
+
raise error_message
|
|
106
|
+
end
|
|
107
|
+
end
|
|
108
|
+
typed_event_data
|
|
109
|
+
end
|
|
110
|
+
|
|
111
|
+
def convert_value_to_cassandra_type(event_data, cassandra_type)
|
|
112
|
+
case cassandra_type
|
|
113
|
+
when 'timestamp'
|
|
114
|
+
converted_value = event_data
|
|
115
|
+
if converted_value.is_a?(Numeric)
|
|
116
|
+
converted_value = Time.at(converted_value)
|
|
117
|
+
elsif converted_value.respond_to?(:to_s)
|
|
118
|
+
converted_value = Time::parse(event_data.to_s)
|
|
119
|
+
end
|
|
120
|
+
return ::Cassandra::Types::Timestamp.new(converted_value)
|
|
121
|
+
when 'inet'
|
|
122
|
+
return ::Cassandra::Types::Inet.new(event_data)
|
|
123
|
+
when 'float'
|
|
124
|
+
return ::Cassandra::Types::Float.new(event_data)
|
|
125
|
+
when 'text'
|
|
126
|
+
return ::Cassandra::Types::Text.new(event_data)
|
|
127
|
+
when 'blob'
|
|
128
|
+
return ::Cassandra::Types::Blob.new(event_data)
|
|
129
|
+
when 'ascii'
|
|
130
|
+
return ::Cassandra::Types::Ascii.new(event_data)
|
|
131
|
+
when 'bigint'
|
|
132
|
+
return ::Cassandra::Types::Bigint.new(event_data)
|
|
133
|
+
when 'counter'
|
|
134
|
+
return ::Cassandra::Types::Counter.new(event_data)
|
|
135
|
+
when 'int'
|
|
136
|
+
return ::Cassandra::Types::Int.new(event_data)
|
|
137
|
+
when 'varint'
|
|
138
|
+
return ::Cassandra::Types::Varint.new(event_data)
|
|
139
|
+
when 'boolean'
|
|
140
|
+
return ::Cassandra::Types::Boolean.new(event_data)
|
|
141
|
+
when 'decimal'
|
|
142
|
+
return ::Cassandra::Types::Decimal.new(event_data)
|
|
143
|
+
when 'double'
|
|
144
|
+
return ::Cassandra::Types::Double.new(event_data)
|
|
145
|
+
when 'timeuuid'
|
|
146
|
+
return ::Cassandra::Types::Timeuuid.new(event_data)
|
|
147
|
+
when 'uuid'
|
|
148
|
+
return ::Cassandra::Types::Uuid.new(event_data)
|
|
149
|
+
when /^set<(.*)>$/
|
|
150
|
+
# convert each value
|
|
151
|
+
# then add all to an array and convert to set
|
|
152
|
+
converted_items = ::Set.new
|
|
153
|
+
set_type = $1
|
|
154
|
+
event_data.each { |item|
|
|
155
|
+
converted_item = convert_value_to_cassandra_type(item, set_type)
|
|
156
|
+
converted_items.add(converted_item)
|
|
157
|
+
}
|
|
158
|
+
return converted_items
|
|
159
|
+
else
|
|
160
|
+
raise "Unknown cassandra_type #{cassandra_type}"
|
|
161
|
+
end
|
|
162
|
+
end
|
|
163
|
+
end
|
|
164
|
+
end end end
|
|
@@ -0,0 +1,121 @@
|
|
|
1
|
+
# encoding: utf-8
|
|
2
|
+
require 'thread'
|
|
3
|
+
require 'cassandra'
|
|
4
|
+
require 'logstash/outputs/cassandra/backoff_retry_policy'
|
|
5
|
+
require 'logstash/outputs/cassandra/schema_fetcher_version_patch'
|
|
6
|
+
|
|
7
|
+
module LogStash; module Outputs; module Cassandra
|
|
8
|
+
# Responsible for submitting parsed actions to cassandra (with or without a retry mechanism)
|
|
9
|
+
class SafeSubmitter
|
|
10
|
+
def initialize(options)
|
|
11
|
+
@statement_cache = {}
|
|
12
|
+
@logger = options['logger']
|
|
13
|
+
setup_cassandra_session(options)
|
|
14
|
+
end
|
|
15
|
+
|
|
16
|
+
def submit(actions)
|
|
17
|
+
queries = prepare_queries(actions)
|
|
18
|
+
execute_queries_with_retries(queries)
|
|
19
|
+
end
|
|
20
|
+
|
|
21
|
+
private
|
|
22
|
+
def setup_cassandra_session(options)
|
|
23
|
+
@retry_policy = get_retry_policy(options['retry_policy'])
|
|
24
|
+
@consistency = options['consistency'].to_sym
|
|
25
|
+
cluster = options['cassandra'].cluster(
|
|
26
|
+
username: options['username'],
|
|
27
|
+
password: options['password'],
|
|
28
|
+
protocol_version: options['protocol_version'],
|
|
29
|
+
hosts: options['hosts'],
|
|
30
|
+
port: options['port'],
|
|
31
|
+
consistency: @consistency,
|
|
32
|
+
timeout: options['request_timeout'],
|
|
33
|
+
retry_policy: @retry_policy,
|
|
34
|
+
logger: options['logger']
|
|
35
|
+
)
|
|
36
|
+
@session = cluster.connect(options['keyspace'])
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
def get_retry_policy(retry_policy)
|
|
40
|
+
case retry_policy['type']
|
|
41
|
+
when 'default'
|
|
42
|
+
return ::Cassandra::Retry::Policies::Default.new
|
|
43
|
+
when 'downgrading_consistency'
|
|
44
|
+
return ::Cassandra::Retry::Policies::DowngradingConsistency.new
|
|
45
|
+
when 'failthrough'
|
|
46
|
+
return ::Cassandra::Retry::Policies::Fallthrough.new
|
|
47
|
+
when 'backoff'
|
|
48
|
+
return ::Cassandra::Retry::Policies::Backoff.new({
|
|
49
|
+
'backoff_type' => retry_policy['backoff_type'], 'backoff_size' => retry_policy['backoff_size'],
|
|
50
|
+
'retry_limit' => retry_policy['retry_limit'], 'logger' => @logger
|
|
51
|
+
})
|
|
52
|
+
else
|
|
53
|
+
raise ArgumentError, "unknown retry policy type: #{retry_policy['type']}"
|
|
54
|
+
end
|
|
55
|
+
end
|
|
56
|
+
|
|
57
|
+
def prepare_queries(actions)
|
|
58
|
+
remaining_queries = Queue.new
|
|
59
|
+
actions.each do |action|
|
|
60
|
+
begin
|
|
61
|
+
if action
|
|
62
|
+
query = get_query(action)
|
|
63
|
+
remaining_queries << { :query => query, :arguments => action['data'].values }
|
|
64
|
+
end
|
|
65
|
+
rescue Exception => e
|
|
66
|
+
@logger.error('Failed to prepare query', :action => action, :exception => e, :backtrace => e.backtrace)
|
|
67
|
+
end
|
|
68
|
+
end
|
|
69
|
+
remaining_queries
|
|
70
|
+
end
|
|
71
|
+
|
|
72
|
+
def get_query(action)
|
|
73
|
+
@logger.debug('generating query for action', :action => action)
|
|
74
|
+
action_data = action['data']
|
|
75
|
+
query =
|
|
76
|
+
"INSERT INTO #{action['table']} (#{action_data.keys.join(', ')})
|
|
77
|
+
VALUES (#{('?' * action_data.keys.count).split(//) * ', '})"
|
|
78
|
+
unless @statement_cache.has_key?(query)
|
|
79
|
+
@logger.debug('preparing new query', :query => query)
|
|
80
|
+
@statement_cache[query] = @session.prepare(query)
|
|
81
|
+
end
|
|
82
|
+
@statement_cache[query]
|
|
83
|
+
end
|
|
84
|
+
|
|
85
|
+
def execute_queries_with_retries(queries)
|
|
86
|
+
retries = 0
|
|
87
|
+
while queries.length > 0
|
|
88
|
+
execute_queries(queries, retries)
|
|
89
|
+
retries += 1
|
|
90
|
+
end
|
|
91
|
+
end
|
|
92
|
+
|
|
93
|
+
def execute_queries(queries, retries)
|
|
94
|
+
futures = []
|
|
95
|
+
while queries.length > 0
|
|
96
|
+
query = queries.pop
|
|
97
|
+
begin
|
|
98
|
+
future = execute_async(query, retries, queries)
|
|
99
|
+
futures << future
|
|
100
|
+
rescue Exception => e
|
|
101
|
+
@logger.error('Failed to send query', :query => query, :exception => e, :backtrace => e.backtrace)
|
|
102
|
+
end
|
|
103
|
+
end
|
|
104
|
+
futures.each(&:join)
|
|
105
|
+
end
|
|
106
|
+
|
|
107
|
+
def execute_async(query, retries, queries)
|
|
108
|
+
future = @session.execute_async(query[:query], arguments: query[:arguments])
|
|
109
|
+
future.on_failure { |error|
|
|
110
|
+
@logger.error('Failed to execute query', :query => query, :error => error)
|
|
111
|
+
if @retry_policy.is_a?(::Cassandra::Retry::Policies::Backoff)
|
|
112
|
+
decision = @retry_policy.retry_with_backoff({ :retries => retries, :consistency => @consistency })
|
|
113
|
+
if decision.is_a?(::Cassandra::Retry::Decisions::Retry)
|
|
114
|
+
queries << query
|
|
115
|
+
end
|
|
116
|
+
end
|
|
117
|
+
}
|
|
118
|
+
future
|
|
119
|
+
end
|
|
120
|
+
end
|
|
121
|
+
end end end
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
# encoding: utf-8
|
|
2
|
+
require 'cassandra'
|
|
3
|
+
|
|
4
|
+
# cassandra-driver 3.2.5 (last released in 2020) only registers schema fetchers for
|
|
5
|
+
# Cassandra release versions '1.2', '2.0', '2.1', '2.2', '3.' and '4.' (see
|
|
6
|
+
# Cassandra::Driver#create_schema_fetcher_picker). Anything else, e.g. Cassandra 5.x,
|
|
7
|
+
# makes the control connection fail with:
|
|
8
|
+
# Cassandra::Errors::ClientError: unsupported release version "5.0.8".
|
|
9
|
+
# The system_schema.* tables that the '3.'/'4.' fetcher (V3_0_x) reads from have not
|
|
10
|
+
# changed in a way that breaks it for Cassandra 5.x, so it's safe to reuse it here.
|
|
11
|
+
module LogStash; module Outputs; module Cassandra
|
|
12
|
+
module SchemaFetcherVersionPatch
|
|
13
|
+
def create_schema_fetcher_picker
|
|
14
|
+
picker = super
|
|
15
|
+
picker.when('5.') do
|
|
16
|
+
::Cassandra::Cluster::Schema::Fetchers::V3_0_x.new(schema_cql_type_parser, cluster_schema)
|
|
17
|
+
end
|
|
18
|
+
picker
|
|
19
|
+
end
|
|
20
|
+
end
|
|
21
|
+
end; end; end
|
|
22
|
+
|
|
23
|
+
::Cassandra::Driver.prepend(LogStash::Outputs::Cassandra::SchemaFetcherVersionPatch)
|