fluentd 1.19.2 → 1.19.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +81 -0
- data/Rakefile +7 -0
- data/lib/fluent/command/cat.rb +1 -1
- data/lib/fluent/config/basic_parser.rb +0 -1
- data/lib/fluent/config/literal_parser.rb +18 -7
- data/lib/fluent/config/types.rb +10 -2
- data/lib/fluent/configurable.rb +1 -0
- data/lib/fluent/daemon.rb +2 -1
- data/lib/fluent/engine.rb +1 -1
- data/lib/fluent/env.rb +1 -0
- data/lib/fluent/event.rb +2 -1
- data/lib/fluent/plugin/base.rb +7 -2
- data/lib/fluent/plugin/buf_file.rb +4 -4
- data/lib/fluent/plugin/buf_file_single.rb +2 -2
- data/lib/fluent/plugin/buf_memory.rb +1 -1
- data/lib/fluent/plugin/buffer/chunk.rb +16 -7
- data/lib/fluent/plugin/buffer/file_chunk.rb +7 -5
- data/lib/fluent/plugin/buffer/file_single_chunk.rb +7 -5
- data/lib/fluent/plugin/buffer/memory_chunk.rb +1 -1
- data/lib/fluent/plugin/buffer.rb +26 -5
- data/lib/fluent/plugin/compressable.rb +7 -62
- data/lib/fluent/plugin/extractor.rb +132 -0
- data/lib/fluent/plugin/filter_record_transformer.rb +2 -1
- data/lib/fluent/plugin/in_debug_agent.rb +1 -1
- data/lib/fluent/plugin/in_forward.rb +3 -1
- data/lib/fluent/plugin/in_http.rb +91 -6
- data/lib/fluent/plugin/in_monitor_agent.rb +21 -23
- data/lib/fluent/plugin/in_sample.rb +2 -1
- data/lib/fluent/plugin/in_syslog.rb +70 -4
- data/lib/fluent/plugin/out_file.rb +10 -0
- data/lib/fluent/plugin/out_forward/connection_manager.rb +18 -0
- data/lib/fluent/plugin/out_forward/socket_cache.rb +45 -10
- data/lib/fluent/plugin/out_forward.rb +13 -2
- data/lib/fluent/plugin/out_http.rb +31 -1
- data/lib/fluent/plugin/output.rb +69 -5
- data/lib/fluent/plugin/parser_csv.rb +5 -0
- data/lib/fluent/plugin/parser_json.rb +6 -1
- data/lib/fluent/plugin/parser_syslog.rb +3 -3
- data/lib/fluent/plugin/sd_file.rb +2 -1
- data/lib/fluent/plugin/storage_local.rb +4 -4
- data/lib/fluent/supervisor.rb +22 -1
- data/lib/fluent/test/base.rb +5 -0
- data/lib/fluent/version.rb +1 -1
- metadata +3 -3
- data/.deepsource.toml +0 -13
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: cab6a9d7ebeaec3be80f888c3b787fb8810fbd9bb78aad85aa5a96192d33365e
|
|
4
|
+
data.tar.gz: d80e36b87699060e8c895d9fb332dc521f0ba9cdbb0b0f4c80dd996003fdca50
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: 2ace1e9bde616fc42f958893bd9b8540f8047981f33bb7e22b8f595bf0296900125d8609976f45e2560f97b4cac40da5420adfa6d25589c0a04115093fb1aa40
|
|
7
|
+
data.tar.gz: 7af35f61b6d84663e4871f410e9f4d32116d59e5953ecf32797fbb3331f5caefff699d1a55820967b4b45d183ac0980a9776284e4d29ed94828264d5648d7992
|
data/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,86 @@
|
|
|
1
1
|
# v1.19
|
|
2
2
|
|
|
3
|
+
## Release v1.19.4 - 2026/09/28
|
|
4
|
+
|
|
5
|
+
### Bug Fix
|
|
6
|
+
|
|
7
|
+
* buffer: enforce decompression_size_limit on the chunk IO path https://github.com/fluent/fluentd/pull/5504
|
|
8
|
+
* buffer: clamp exported buffer size metrics to non-negative values https://github.com/fluent/fluentd/pull/5488
|
|
9
|
+
* buffer: fix spurious `BufferOverflowError` caused by `queue_size` leaking when a chunk purge fails https://github.com/fluent/fluentd/pull/5487
|
|
10
|
+
* buffer: fix `stage_byte_size` leak on staged to unstaged chunk demotion that could eventually raise spurious `BufferOverflowError` in plugins implementing `#format` https://github.com/fluent/fluentd/pull/5456
|
|
11
|
+
* in_syslog: enforce message_length_limit on TCP/TLS transport https://github.com/fluent/fluentd/pull/5502
|
|
12
|
+
* The default value of `message_length_limit` is changed from 2048 to 8192 to match rsyslog's default `MaxMessageSize`.
|
|
13
|
+
* output: fix incomplete path traversal check in extract_placeholders https://github.com/fluent/fluentd/pull/5501
|
|
14
|
+
* output: treat JSON::GeneratorError as unrecoverable error https://github.com/fluent/fluentd/pull/5423
|
|
15
|
+
* parser_syslog: fix NameError when RFC5424 timestamp has repeated spaces https://github.com/fluent/fluentd/pull/5500
|
|
16
|
+
* parser_syslog: fix NameError when RFC3164 timestamp has repeated spaces https://github.com/fluent/fluentd/pull/5497
|
|
17
|
+
* parser_syslog: avoid excessive backtracking when parsing malformed RFC5424 structured data https://github.com/fluent/fluentd/pull/5444
|
|
18
|
+
* chunk: ensure to close the Tempfile for decompressed data https://github.com/fluent/fluentd/pull/5486
|
|
19
|
+
* plugin base: bound the number of worker lock files by hashing the path into a fixed set of buckets https://github.com/fluent/fluentd/pull/5471
|
|
20
|
+
* out_forward: stop the endless "ack in response and chunk id in sent data are different" warning storm by discarding (instead of reusing) a keepalive socket whose ack failed or came back with a mismatched chunk id https://github.com/fluent/fluentd/pull/5445
|
|
21
|
+
* config: accept empty lines in quoted strings https://github.com/fluent/fluentd/pull/5478
|
|
22
|
+
* config: fix a config error when a single scalar value is given to an array option in YAML config syntax (for example `retryable_response_codes: 503`) https://github.com/fluent/fluentd/pull/5433
|
|
23
|
+
* supervisor: reduce memory usage of cleanup_lock_dir with huge number of lock files https://github.com/fluent/fluentd/pull/5472
|
|
24
|
+
|
|
25
|
+
### Enhancement
|
|
26
|
+
|
|
27
|
+
* in_http: add `<auth>` for basic authentication and `<security>` for client network allowlisting https://github.com/fluent/fluentd/pull/5503
|
|
28
|
+
|
|
29
|
+
### Misc
|
|
30
|
+
|
|
31
|
+
* gem: support json gem v3.x https://github.com/fluent/fluentd/pull/5493
|
|
32
|
+
* Add `allow_comments: true` option for json parser https://github.com/fluent/fluentd/pull/5432
|
|
33
|
+
* Set `allow_duplicate_key: true` for all `JSON.parse` call https://github.com/fluent/fluentd/pull/5430
|
|
34
|
+
* CI fixes
|
|
35
|
+
* https://github.com/fluent/fluentd/pull/5495
|
|
36
|
+
* https://github.com/fluent/fluentd/pull/5477
|
|
37
|
+
* https://github.com/fluent/fluentd/pull/5466
|
|
38
|
+
* https://github.com/fluent/fluentd/pull/5460
|
|
39
|
+
* https://github.com/fluent/fluentd/pull/5455
|
|
40
|
+
* https://github.com/fluent/fluentd/pull/5454
|
|
41
|
+
* https://github.com/fluent/fluentd/pull/5453
|
|
42
|
+
* https://github.com/fluent/fluentd/pull/5452
|
|
43
|
+
* https://github.com/fluent/fluentd/pull/5434
|
|
44
|
+
* https://github.com/fluent/fluentd/pull/5426
|
|
45
|
+
* https://github.com/fluent/fluentd/pull/5425
|
|
46
|
+
* https://github.com/fluent/fluentd/pull/5424
|
|
47
|
+
|
|
48
|
+
## Release v1.19.3 - 2026/06/25
|
|
49
|
+
|
|
50
|
+
### Bug Fix
|
|
51
|
+
|
|
52
|
+
* out_http: add strict host validation for dynamic endpoints https://github.com/fluent/fluentd/pull/5394
|
|
53
|
+
* buffer, in_http: enforce size limits on decompressed payloads https://github.com/fluent/fluentd/pull/5393
|
|
54
|
+
* in_monitor_agent: change default visibility of config, retry, and debug info https://github.com/fluent/fluentd/pull/5392
|
|
55
|
+
* output: enforce strict path boundary validation for tag https://github.com/fluent/fluentd/pull/5391
|
|
56
|
+
* engine: remove duplicated word in unreloadable plugin error message https://github.com/fluent/fluentd/pull/5389
|
|
57
|
+
* storage_local: fix encoding error when reading non-ASCII characters https://github.com/fluent/fluentd/pull/5382
|
|
58
|
+
* parser_csv: skip empty or unparseable lines https://github.com/fluent/fluentd/pull/5359
|
|
59
|
+
* out_forward: avoid reusing closed keepalive sockets after remote disconnects https://github.com/fluent/fluentd/pull/5343
|
|
60
|
+
* buffer: resume buffer correctly even though path contains [] https://github.com/fluent/fluentd/pull/5305
|
|
61
|
+
* in_debug_agent: accept only from local machine by default https://github.com/fluent/fluentd/pull/5279
|
|
62
|
+
|
|
63
|
+
### Misc
|
|
64
|
+
|
|
65
|
+
* gem: add win32-registry as runtime dependency for Ruby 4.1 https://github.com/fluent/fluentd/pull/5317
|
|
66
|
+
* output windows: check shorter service timeout on shutdown https://github.com/fluent/fluentd/pull/5306
|
|
67
|
+
* buffer: warn if default timekey (1d) will be used https://github.com/fluent/fluentd/pull/5291
|
|
68
|
+
* warn recommended exclusion path for antivirus https://github.com/fluent/fluentd/pull/5280
|
|
69
|
+
* CI fixes
|
|
70
|
+
* https://github.com/fluent/fluentd/pull/5387
|
|
71
|
+
* https://github.com/fluent/fluentd/pull/5386
|
|
72
|
+
* https://github.com/fluent/fluentd/pull/5366
|
|
73
|
+
* https://github.com/fluent/fluentd/pull/5342
|
|
74
|
+
* https://github.com/fluent/fluentd/pull/5341
|
|
75
|
+
* https://github.com/fluent/fluentd/pull/5340
|
|
76
|
+
* https://github.com/fluent/fluentd/pull/5339
|
|
77
|
+
* https://github.com/fluent/fluentd/pull/5338
|
|
78
|
+
* https://github.com/fluent/fluentd/pull/5337
|
|
79
|
+
* https://github.com/fluent/fluentd/pull/5336
|
|
80
|
+
* https://github.com/fluent/fluentd/pull/5335
|
|
81
|
+
* https://github.com/fluent/fluentd/pull/5334
|
|
82
|
+
* https://github.com/fluent/fluentd/pull/5333
|
|
83
|
+
|
|
3
84
|
## Release v1.19.2 - 2026/02/13
|
|
4
85
|
|
|
5
86
|
### Bug Fix
|
data/Rakefile
CHANGED
|
@@ -78,4 +78,11 @@ task :coverity do
|
|
|
78
78
|
FileUtils.rm_rf(['./cov-int', 'cov-fluentd.tar.gz'])
|
|
79
79
|
end
|
|
80
80
|
|
|
81
|
+
task :check_env do
|
|
82
|
+
unless ENV['GEM_HOST_API_KEY']
|
|
83
|
+
abort "Missing required environment variable: GEM_HOST_API_KEY\nSee https://guides.rubygems.org/api-key-scopes/"
|
|
84
|
+
end
|
|
85
|
+
end
|
|
86
|
+
Rake::Task["release:rubygem_push"].enhance(["check_env"])
|
|
87
|
+
|
|
81
88
|
task default: [:test, :build]
|
data/lib/fluent/command/cat.rb
CHANGED
|
@@ -28,7 +28,6 @@ module Fluent
|
|
|
28
28
|
SPACING = /(?:[ \t\r\n]|\z|\#.*?(?:\z|[\r\n]))+/
|
|
29
29
|
ZERO_OR_MORE_SPACING = /(?:[ \t\r\n]|\z|\#.*?(?:\z|[\r\n]))*/
|
|
30
30
|
SPACING_WITHOUT_COMMENT = /(?:[ \t\r\n]|\z)+/
|
|
31
|
-
LINE_END_WITHOUT_SPACING_AND_COMMENT = /(?:\z|[\r\n])/
|
|
32
31
|
|
|
33
32
|
module ClassMethods
|
|
34
33
|
def symbol(string)
|
|
@@ -21,11 +21,21 @@ require 'yajl'
|
|
|
21
21
|
require 'socket'
|
|
22
22
|
require 'ripper'
|
|
23
23
|
|
|
24
|
+
require 'fluent/env'
|
|
24
25
|
require 'fluent/config/basic_parser'
|
|
25
26
|
|
|
26
27
|
module Fluent
|
|
27
28
|
module Config
|
|
28
29
|
class LiteralParser < BasicParser
|
|
30
|
+
# A physical line break (LF, CR, or CRLF) inside a quoted string.
|
|
31
|
+
# CRLF is normalized to LF so that the same config text does not produce a
|
|
32
|
+
# different value depending on whether the file was saved with LF or CRLF.
|
|
33
|
+
# A lone CR is kept as-is, and an escaped "\r\n" still produces CRLF.
|
|
34
|
+
LINE_BREAK = /\r\n|[\r\n]/
|
|
35
|
+
# A backslash immediately followed by a physical line break.
|
|
36
|
+
# It works as a line continuation, so both are stripped from the value.
|
|
37
|
+
LINE_CONTINUATION = /\\#{LINE_BREAK}/o
|
|
38
|
+
|
|
29
39
|
def self.unescape_char(c)
|
|
30
40
|
case c
|
|
31
41
|
when '"'
|
|
@@ -98,11 +108,10 @@ module Fluent
|
|
|
98
108
|
else
|
|
99
109
|
return string.join
|
|
100
110
|
end
|
|
101
|
-
elsif
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
skip(/[^"]#{LINE_END_WITHOUT_SPACING_AND_COMMENT}/o)
|
|
111
|
+
elsif skip(LINE_CONTINUATION)
|
|
112
|
+
next
|
|
113
|
+
elsif s = scan(LINE_BREAK)
|
|
114
|
+
string << (s == "\r\n" ? "\n" : s)
|
|
106
115
|
elsif s = scan(/\\./)
|
|
107
116
|
string << eval_escape_char(s[1,1])
|
|
108
117
|
elsif skip(/\#\{/)
|
|
@@ -125,6 +134,8 @@ module Fluent
|
|
|
125
134
|
string << "'"
|
|
126
135
|
elsif s = scan(/\\\\/)
|
|
127
136
|
string << "\\"
|
|
137
|
+
elsif s = scan(LINE_BREAK)
|
|
138
|
+
string << (s == "\r\n" ? "\n" : s)
|
|
128
139
|
elsif s = scan(/./)
|
|
129
140
|
string << s
|
|
130
141
|
else
|
|
@@ -241,7 +252,7 @@ EOM
|
|
|
241
252
|
# '{"foo":"bar", #' -> '{"foo":"bar"}' (to check)
|
|
242
253
|
parsed = nil
|
|
243
254
|
begin
|
|
244
|
-
parsed = JSON.parse(buffer + line_buffer.rstrip.sub(/,$/, '') + (is_array ? "]" : "}"))
|
|
255
|
+
parsed = JSON.parse(buffer + line_buffer.rstrip.sub(/,$/, '') + (is_array ? "]" : "}"), **Fluent::DEFAULT_JSON_PARSE_OPTIONS)
|
|
245
256
|
rescue JSON::ParserError
|
|
246
257
|
# This '#' is in json string literals
|
|
247
258
|
end
|
|
@@ -276,7 +287,7 @@ EOM
|
|
|
276
287
|
|
|
277
288
|
line_buffer << char
|
|
278
289
|
begin
|
|
279
|
-
result = JSON.parse(buffer + line_buffer)
|
|
290
|
+
result = JSON.parse(buffer + line_buffer, **Fluent::DEFAULT_JSON_PARSE_OPTIONS)
|
|
280
291
|
rescue JSON::ParserError
|
|
281
292
|
# Incomplete json string yet
|
|
282
293
|
end
|
data/lib/fluent/config/types.rb
CHANGED
|
@@ -201,7 +201,7 @@ module Fluent
|
|
|
201
201
|
return nil if val.nil?
|
|
202
202
|
|
|
203
203
|
param = if val.is_a?(String)
|
|
204
|
-
val.start_with?('{') ? JSON.parse(val) : Hash[val.strip.split(/\s*,\s*/).map{|v| v.split(':', 2)}]
|
|
204
|
+
val.start_with?('{') ? JSON.parse(val, **Fluent::DEFAULT_JSON_PARSE_OPTIONS) : Hash[val.strip.split(/\s*,\s*/).map{|v| v.split(':', 2)}]
|
|
205
205
|
else
|
|
206
206
|
val
|
|
207
207
|
end
|
|
@@ -228,7 +228,15 @@ module Fluent
|
|
|
228
228
|
return nil if val.nil?
|
|
229
229
|
|
|
230
230
|
param = if val.is_a?(String)
|
|
231
|
-
val.start_with?('[') ? JSON.parse(val) : val.strip.split(/\s*,\s*/)
|
|
231
|
+
val.start_with?('[') ? JSON.parse(val, **Fluent::DEFAULT_JSON_PARSE_OPTIONS) : val.strip.split(/\s*,\s*/)
|
|
232
|
+
elsif val.is_a?(Array)
|
|
233
|
+
val
|
|
234
|
+
elsif val.is_a?(Numeric) || val == true || val == false
|
|
235
|
+
# Wrap only the bare scalars that Psych/YAML actually produces
|
|
236
|
+
# here (nil is handled by the early return above). Any other
|
|
237
|
+
# type (Hash, Symbol, Time, arbitrary objects) falls through and
|
|
238
|
+
# still raises "array required" below.
|
|
239
|
+
[val]
|
|
232
240
|
else
|
|
233
241
|
val
|
|
234
242
|
end
|
data/lib/fluent/configurable.rb
CHANGED
data/lib/fluent/daemon.rb
CHANGED
|
@@ -5,9 +5,10 @@ here = File.dirname(__FILE__)
|
|
|
5
5
|
$LOAD_PATH << File.expand_path(File.join(here, '..'))
|
|
6
6
|
|
|
7
7
|
require 'serverengine'
|
|
8
|
+
require 'fluent/env'
|
|
8
9
|
require 'fluent/supervisor'
|
|
9
10
|
|
|
10
11
|
server_module = Fluent.const_get(ARGV[0])
|
|
11
12
|
worker_module = Fluent.const_get(ARGV[1])
|
|
12
|
-
params = JSON.parse(ARGV[2])
|
|
13
|
+
params = JSON.parse(ARGV[2], **Fluent::DEFAULT_JSON_PARSE_OPTIONS)
|
|
13
14
|
ServerEngine::Daemon.run_server(server_module, worker_module) { Fluent::Supervisor.serverengine_config(params) }
|
data/lib/fluent/engine.rb
CHANGED
|
@@ -186,7 +186,7 @@ module Fluent
|
|
|
186
186
|
|
|
187
187
|
ret.all_plugins.each do |plugin|
|
|
188
188
|
if plugin.respond_to?(:reloadable_plugin?) && !plugin.reloadable_plugin?
|
|
189
|
-
raise Fluent::ConfigError, "Unreloadable plugin
|
|
189
|
+
raise Fluent::ConfigError, "Unreloadable plugin: #{Fluent::Plugin.lookup_type_from_class(plugin.class)}, plugin_id: #{plugin.plugin_id}, class_name: #{plugin.class})"
|
|
190
190
|
end
|
|
191
191
|
end
|
|
192
192
|
|
data/lib/fluent/env.rb
CHANGED
|
@@ -26,6 +26,7 @@ module Fluent
|
|
|
26
26
|
DEFAULT_SOCKET_PATH = ENV['FLUENT_SOCKET'] || '/var/run/fluent/fluent.sock'
|
|
27
27
|
DEFAULT_BACKUP_DIR = ENV['FLUENT_BACKUP_DIR'] || '/tmp/fluent'
|
|
28
28
|
DEFAULT_OJ_OPTIONS = Fluent::OjOptions.load_env
|
|
29
|
+
DEFAULT_JSON_PARSE_OPTIONS = { allow_duplicate_key: true, allow_comments: true }.freeze
|
|
29
30
|
DEFAULT_DIR_PERMISSION = 0755
|
|
30
31
|
DEFAULT_FILE_PERMISSION = 0644
|
|
31
32
|
INSTANCE_ID = ENV['FLUENT_INSTANCE_ID'] || SecureRandom.uuid
|
data/lib/fluent/event.rb
CHANGED
|
@@ -268,11 +268,12 @@ module Fluent
|
|
|
268
268
|
end
|
|
269
269
|
|
|
270
270
|
class CompressedMessagePackEventStream < MessagePackEventStream
|
|
271
|
-
def initialize(data, cached_unpacker = nil, size = 0, unpacked_times: nil, unpacked_records: nil, compress: :gzip)
|
|
271
|
+
def initialize(data, cached_unpacker = nil, size = 0, unpacked_times: nil, unpacked_records: nil, compress: :gzip, decompression_size_limit: Fluent::Plugin::Compressable::DEFAULT_DECOMPRESSION_SIZE_LIMIT)
|
|
272
272
|
super(data, cached_unpacker, size, unpacked_times: unpacked_times, unpacked_records: unpacked_records)
|
|
273
273
|
@decompressed_data = nil
|
|
274
274
|
@compressed_data = data
|
|
275
275
|
@type = compress
|
|
276
|
+
@decompression_size_limit = decompression_size_limit
|
|
276
277
|
end
|
|
277
278
|
|
|
278
279
|
def empty?
|
data/lib/fluent/plugin/base.rb
CHANGED
|
@@ -14,6 +14,7 @@
|
|
|
14
14
|
# limitations under the License.
|
|
15
15
|
#
|
|
16
16
|
|
|
17
|
+
require 'zlib'
|
|
17
18
|
require 'fluent/plugin'
|
|
18
19
|
require 'fluent/configurable'
|
|
19
20
|
require 'fluent/system_config'
|
|
@@ -69,9 +70,13 @@ module Fluent
|
|
|
69
70
|
true
|
|
70
71
|
end
|
|
71
72
|
|
|
73
|
+
LOCK_FILE_BUCKETS = 65536
|
|
74
|
+
|
|
72
75
|
def get_lock_path(name)
|
|
73
|
-
name
|
|
74
|
-
|
|
76
|
+
# The mapping from a name to a bucket MUST be identical across worker processes.
|
|
77
|
+
# Ruby's String#hash is randomly seeded per process and MUST NOT be used here.
|
|
78
|
+
bucket = Zlib.crc32(name.to_s) % LOCK_FILE_BUCKETS
|
|
79
|
+
File.join(@fluentd_lock_dir, "fluentd-bucket-#{bucket}.lock")
|
|
75
80
|
end
|
|
76
81
|
|
|
77
82
|
def acquire_worker_lock(name)
|
|
@@ -163,7 +163,7 @@ module Fluent
|
|
|
163
163
|
end
|
|
164
164
|
|
|
165
165
|
begin
|
|
166
|
-
chunk = Fluent::Plugin::Buffer::FileChunk.new(m, path, mode, compress: @compress) # file chunk resumes contents of metadata
|
|
166
|
+
chunk = Fluent::Plugin::Buffer::FileChunk.new(m, path, mode, compress: @compress, decompression_size_limit: @decompression_size_limit) # file chunk resumes contents of metadata
|
|
167
167
|
rescue Fluent::Plugin::Buffer::FileChunk::FileChunkError => e
|
|
168
168
|
exist_broken_file = true
|
|
169
169
|
handle_broken_files(path, mode, e)
|
|
@@ -205,7 +205,7 @@ module Fluent
|
|
|
205
205
|
def generate_chunk(metadata)
|
|
206
206
|
# FileChunk generates real path with unique_id
|
|
207
207
|
perm = @file_permission || system_config.file_permission
|
|
208
|
-
chunk = Fluent::Plugin::Buffer::FileChunk.new(metadata, @path, :create, perm: perm, compress: @compress)
|
|
208
|
+
chunk = Fluent::Plugin::Buffer::FileChunk.new(metadata, @path, :create, perm: perm, compress: @compress, decompression_size_limit: @decompression_size_limit)
|
|
209
209
|
log.debug "Created new chunk", chunk_id: dump_unique_id_hex(chunk.unique_id), metadata: metadata
|
|
210
210
|
|
|
211
211
|
return chunk
|
|
@@ -247,8 +247,8 @@ module Fluent
|
|
|
247
247
|
|
|
248
248
|
def escaped_patterns(patterns)
|
|
249
249
|
patterns.map { |pattern|
|
|
250
|
-
# '{' '}' are special character in Dir.glob
|
|
251
|
-
pattern.gsub(/[\{\}]/) { |c| "\\#{c}" }
|
|
250
|
+
# '{', '}', '[' and ']' are special character in Dir.glob
|
|
251
|
+
pattern.gsub(/[\{\}\[\]]/) { |c| "\\#{c}" }
|
|
252
252
|
}
|
|
253
253
|
end
|
|
254
254
|
end
|
|
@@ -183,7 +183,7 @@ module Fluent
|
|
|
183
183
|
end
|
|
184
184
|
|
|
185
185
|
begin
|
|
186
|
-
chunk = Fluent::Plugin::Buffer::FileSingleChunk.new(m, path, mode, @key_in_path, compress: @compress)
|
|
186
|
+
chunk = Fluent::Plugin::Buffer::FileSingleChunk.new(m, path, mode, @key_in_path, compress: @compress, decompression_size_limit: @decompression_size_limit)
|
|
187
187
|
chunk.restore_size(@chunk_format) if @calc_num_records
|
|
188
188
|
rescue Fluent::Plugin::Buffer::FileSingleChunk::FileChunkError => e
|
|
189
189
|
exist_broken_file = true
|
|
@@ -216,7 +216,7 @@ module Fluent
|
|
|
216
216
|
def generate_chunk(metadata)
|
|
217
217
|
# FileChunk generates real path with unique_id
|
|
218
218
|
perm = @file_permission || system_config.file_permission
|
|
219
|
-
chunk = Fluent::Plugin::Buffer::FileSingleChunk.new(metadata, @path, :create, @key_in_path, perm: perm, compress: @compress)
|
|
219
|
+
chunk = Fluent::Plugin::Buffer::FileSingleChunk.new(metadata, @path, :create, @key_in_path, perm: perm, compress: @compress, decompression_size_limit: @decompression_size_limit)
|
|
220
220
|
|
|
221
221
|
log.debug "Created new chunk", chunk_id: dump_unique_id_hex(chunk.unique_id), metadata: metadata
|
|
222
222
|
|
|
@@ -27,7 +27,7 @@ module Fluent
|
|
|
27
27
|
end
|
|
28
28
|
|
|
29
29
|
def generate_chunk(metadata)
|
|
30
|
-
Fluent::Plugin::Buffer::MemoryChunk.new(metadata, compress: @compress)
|
|
30
|
+
Fluent::Plugin::Buffer::MemoryChunk.new(metadata, compress: @compress, decompression_size_limit: @decompression_size_limit)
|
|
31
31
|
end
|
|
32
32
|
end
|
|
33
33
|
end
|
|
@@ -48,7 +48,7 @@ module Fluent
|
|
|
48
48
|
|
|
49
49
|
# TODO: CompressedPackedMessage of forward protocol?
|
|
50
50
|
|
|
51
|
-
def initialize(metadata, compress: :text)
|
|
51
|
+
def initialize(metadata, compress: :text, decompression_size_limit: Compressable::DEFAULT_DECOMPRESSION_SIZE_LIMIT)
|
|
52
52
|
super()
|
|
53
53
|
@unique_id = generate_unique_id
|
|
54
54
|
@metadata = metadata
|
|
@@ -64,6 +64,7 @@ module Fluent
|
|
|
64
64
|
elsif compress == :zstd
|
|
65
65
|
extend ZstdDecompressable
|
|
66
66
|
end
|
|
67
|
+
@decompression_size_limit = decompression_size_limit
|
|
67
68
|
end
|
|
68
69
|
|
|
69
70
|
attr_reader :unique_id, :metadata, :state
|
|
@@ -219,9 +220,13 @@ module Fluent
|
|
|
219
220
|
Tempfile.new('decompressed-data')
|
|
220
221
|
end
|
|
221
222
|
output_io.binmode if output_io.is_a?(Tempfile)
|
|
222
|
-
|
|
223
|
-
|
|
224
|
-
|
|
223
|
+
begin
|
|
224
|
+
decompress(input_io: chunk_io, output_io: output_io)
|
|
225
|
+
output_io.seek(0, IO::SEEK_SET)
|
|
226
|
+
yield output_io
|
|
227
|
+
ensure
|
|
228
|
+
output_io.close! if output_io.is_a?(Tempfile)
|
|
229
|
+
end
|
|
225
230
|
end
|
|
226
231
|
end
|
|
227
232
|
end
|
|
@@ -273,9 +278,13 @@ module Fluent
|
|
|
273
278
|
Tempfile.new('decompressed-data')
|
|
274
279
|
end
|
|
275
280
|
output_io.binmode if output_io.is_a?(Tempfile)
|
|
276
|
-
|
|
277
|
-
|
|
278
|
-
|
|
281
|
+
begin
|
|
282
|
+
decompress(input_io: chunk_io, output_io: output_io, type: :zstd)
|
|
283
|
+
output_io.seek(0, IO::SEEK_SET)
|
|
284
|
+
yield output_io
|
|
285
|
+
ensure
|
|
286
|
+
output_io.close! if output_io.is_a?(Tempfile)
|
|
287
|
+
end
|
|
279
288
|
end
|
|
280
289
|
end
|
|
281
290
|
end
|
|
@@ -39,8 +39,8 @@ module Fluent
|
|
|
39
39
|
|
|
40
40
|
attr_reader :path, :meta_path, :permission
|
|
41
41
|
|
|
42
|
-
def initialize(metadata, path, mode, perm: nil, compress: :text)
|
|
43
|
-
super(metadata, compress: compress)
|
|
42
|
+
def initialize(metadata, path, mode, perm: nil, compress: :text, decompression_size_limit: Compressable::DEFAULT_DECOMPRESSION_SIZE_LIMIT)
|
|
43
|
+
super(metadata, compress: compress, decompression_size_limit: decompression_size_limit)
|
|
44
44
|
perm ||= Fluent::DEFAULT_FILE_PERMISSION
|
|
45
45
|
@permission = perm.is_a?(String) ? perm.to_i(8) : perm
|
|
46
46
|
@bytesize = @size = @adding_bytes = @adding_size = 0
|
|
@@ -168,9 +168,11 @@ module Fluent
|
|
|
168
168
|
|
|
169
169
|
def open(**kwargs, &block)
|
|
170
170
|
@chunk.seek(0, IO::SEEK_SET)
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
171
|
+
begin
|
|
172
|
+
yield @chunk
|
|
173
|
+
ensure
|
|
174
|
+
@chunk.seek(0, IO::SEEK_END) if self.staged?
|
|
175
|
+
end
|
|
174
176
|
end
|
|
175
177
|
|
|
176
178
|
def self.assume_chunk_state(path)
|
|
@@ -35,8 +35,8 @@ module Fluent
|
|
|
35
35
|
|
|
36
36
|
attr_reader :path, :permission
|
|
37
37
|
|
|
38
|
-
def initialize(metadata, path, mode, key, perm: Fluent::DEFAULT_FILE_PERMISSION, compress: :text)
|
|
39
|
-
super(metadata, compress: compress)
|
|
38
|
+
def initialize(metadata, path, mode, key, perm: Fluent::DEFAULT_FILE_PERMISSION, compress: :text, decompression_size_limit: Compressable::DEFAULT_DECOMPRESSION_SIZE_LIMIT)
|
|
39
|
+
super(metadata, compress: compress, decompression_size_limit: decompression_size_limit)
|
|
40
40
|
@key = key
|
|
41
41
|
perm ||= Fluent::DEFAULT_FILE_PERMISSION
|
|
42
42
|
@permission = perm.is_a?(String) ? perm.to_i(8) : perm
|
|
@@ -141,9 +141,11 @@ module Fluent
|
|
|
141
141
|
|
|
142
142
|
def open(**kwargs, &block)
|
|
143
143
|
@chunk.seek(0, IO::SEEK_SET)
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
144
|
+
begin
|
|
145
|
+
yield @chunk
|
|
146
|
+
ensure
|
|
147
|
+
@chunk.seek(0, IO::SEEK_END) if self.staged?
|
|
148
|
+
end
|
|
147
149
|
end
|
|
148
150
|
|
|
149
151
|
def self.assume_chunk_state(path)
|
|
@@ -20,7 +20,7 @@ module Fluent
|
|
|
20
20
|
module Plugin
|
|
21
21
|
class Buffer
|
|
22
22
|
class MemoryChunk < Chunk
|
|
23
|
-
def initialize(metadata, compress: :text)
|
|
23
|
+
def initialize(metadata, compress: :text, decompression_size_limit: Compressable::DEFAULT_DECOMPRESSION_SIZE_LIMIT)
|
|
24
24
|
super
|
|
25
25
|
@chunk = ''.force_encoding(Encoding::ASCII_8BIT)
|
|
26
26
|
@chunk_bytes = 0
|
data/lib/fluent/plugin/buffer.rb
CHANGED
|
@@ -66,6 +66,9 @@ module Fluent
|
|
|
66
66
|
desc 'Compress buffered data.'
|
|
67
67
|
config_param :compress, :enum, list: [:text, :gzip, :zstd], default: :text
|
|
68
68
|
|
|
69
|
+
desc 'The size limit of the decompressed element.'
|
|
70
|
+
config_param :decompression_size_limit, :size, default: Compressable::DEFAULT_DECOMPRESSION_SIZE_LIMIT
|
|
71
|
+
|
|
69
72
|
desc 'If true, chunks are thrown away when unrecoverable error happens'
|
|
70
73
|
config_param :disable_chunk_backup, :bool, default: false
|
|
71
74
|
|
|
@@ -598,13 +601,19 @@ module Fluent
|
|
|
598
601
|
metadata = chunk.metadata
|
|
599
602
|
log.on_trace { log.trace "purging a chunk", instance: self.object_id, chunk_id: dump_unique_id_hex(chunk_id), metadata: metadata }
|
|
600
603
|
|
|
604
|
+
bytesize = chunk.bytesize
|
|
601
605
|
begin
|
|
602
|
-
bytesize = chunk.bytesize
|
|
603
606
|
chunk.purge
|
|
604
|
-
@queue_size_metrics.sub(bytesize)
|
|
605
607
|
rescue => e
|
|
606
608
|
log.error "failed to purge buffer chunk", chunk_id: dump_unique_id_hex(chunk_id), error_class: e.class, error: e
|
|
607
609
|
log.error_backtrace
|
|
610
|
+
ensure
|
|
611
|
+
# Always release the queued byte counter, even when purge raises
|
|
612
|
+
# (e.g. unlink/close failure on the buffer path). Otherwise
|
|
613
|
+
# @queue_size is never decremented for a dequeued chunk that is
|
|
614
|
+
# gone from both @queue and @dequeued, so the leak ratchets toward
|
|
615
|
+
# total_limit_size and storable? becomes permanently false (#5468).
|
|
616
|
+
@queue_size_metrics.sub(bytesize)
|
|
608
617
|
end
|
|
609
618
|
|
|
610
619
|
@dequeued_num[chunk.metadata] -= 1
|
|
@@ -847,7 +856,12 @@ module Fluent
|
|
|
847
856
|
# As already processed content is kept after rollback, then unstaged chunk should be queued.
|
|
848
857
|
# After that, re-process current split again.
|
|
849
858
|
# New chunk should be allocated, to do it, modify @stage and so on.
|
|
850
|
-
synchronize
|
|
859
|
+
synchronize do
|
|
860
|
+
@stage.delete(modified_metadata)
|
|
861
|
+
# Subtract this chunk's already-counted staged bytes here;
|
|
862
|
+
# otherwise @stage_size leaks upward over the buffer's lifetime.
|
|
863
|
+
@stage_size_metrics.sub(original_bytesize) if chunk.staged?
|
|
864
|
+
end
|
|
851
865
|
staged_chunk_used = false
|
|
852
866
|
chunk.unstaged!
|
|
853
867
|
break
|
|
@@ -909,8 +923,15 @@ module Fluent
|
|
|
909
923
|
]
|
|
910
924
|
|
|
911
925
|
def statistics
|
|
912
|
-
|
|
913
|
-
|
|
926
|
+
# Export-only clamp: internal gauges may go transiently negative during
|
|
927
|
+
# the deferred stage_size add vs enqueue_chunk sub race (#5303, #2712).
|
|
928
|
+
# Clamping the gauge store itself would turn that into a permanent
|
|
929
|
+
# over-count and break Buffer#storable? -- keep raw gauge semantics.
|
|
930
|
+
stage_size = [@stage_size_metrics.get, 0].max
|
|
931
|
+
queue_size = [@queue_size_metrics.get, 0].max
|
|
932
|
+
denom = @total_limit_size.to_f
|
|
933
|
+
# denom > 0 already excludes 0/0 NaN; stage/queue are floored above.
|
|
934
|
+
buffer_space = denom > 0.0 ? (1.0 - (stage_size + queue_size).to_f / denom).clamp(0.0, 1.0) : 0.0
|
|
914
935
|
@stage_length_metrics.set(@stage.size)
|
|
915
936
|
@queue_length_metrics.set(@queue.size)
|
|
916
937
|
@available_buffer_space_ratios_metrics.set(buffer_space * 100)
|
|
@@ -14,6 +14,7 @@
|
|
|
14
14
|
# limitations under the License.
|
|
15
15
|
#
|
|
16
16
|
|
|
17
|
+
require 'fluent/plugin/extractor'
|
|
17
18
|
require 'stringio'
|
|
18
19
|
require 'zlib'
|
|
19
20
|
require 'zstd-ruby'
|
|
@@ -21,6 +22,8 @@ require 'zstd-ruby'
|
|
|
21
22
|
module Fluent
|
|
22
23
|
module Plugin
|
|
23
24
|
module Compressable
|
|
25
|
+
DEFAULT_DECOMPRESSION_SIZE_LIMIT = 256 * 1024 * 1024
|
|
26
|
+
|
|
24
27
|
def compress(data, type: :gzip, **kwargs)
|
|
25
28
|
output_io = kwargs[:output_io]
|
|
26
29
|
io = output_io || StringIO.new
|
|
@@ -60,79 +63,21 @@ module Fluent
|
|
|
60
63
|
|
|
61
64
|
private
|
|
62
65
|
|
|
63
|
-
def string_decompress_gzip(compressed_data)
|
|
64
|
-
io = StringIO.new(compressed_data)
|
|
65
|
-
out = ''
|
|
66
|
-
loop do
|
|
67
|
-
reader = Zlib::GzipReader.new(io)
|
|
68
|
-
out << reader.read
|
|
69
|
-
unused = reader.unused
|
|
70
|
-
reader.finish
|
|
71
|
-
unless unused.nil?
|
|
72
|
-
adjust = unused.length
|
|
73
|
-
io.pos -= adjust
|
|
74
|
-
end
|
|
75
|
-
break if io.eof?
|
|
76
|
-
end
|
|
77
|
-
out
|
|
78
|
-
end
|
|
79
|
-
|
|
80
|
-
def string_decompress_zstd(compressed_data)
|
|
81
|
-
io = StringIO.new(compressed_data)
|
|
82
|
-
reader = Zstd::StreamReader.new(io)
|
|
83
|
-
out = ''
|
|
84
|
-
loop do
|
|
85
|
-
# Zstd::StreamReader needs to specify the size of the buffer
|
|
86
|
-
out << reader.read(1024)
|
|
87
|
-
# Zstd::StreamReader doesn't provide unused data, so we have to manually adjust the position
|
|
88
|
-
break if io.eof?
|
|
89
|
-
end
|
|
90
|
-
out
|
|
91
|
-
end
|
|
92
|
-
|
|
93
66
|
def string_decompress(compressed_data, type = :gzip)
|
|
94
67
|
if type == :gzip
|
|
95
|
-
|
|
68
|
+
Extractor.decompress_gzip(compressed_data, limit: @decompression_size_limit || DEFAULT_DECOMPRESSION_SIZE_LIMIT)
|
|
96
69
|
elsif type == :zstd
|
|
97
|
-
|
|
70
|
+
Extractor.decompress_zstd(compressed_data, limit: @decompression_size_limit || DEFAULT_DECOMPRESSION_SIZE_LIMIT)
|
|
98
71
|
else
|
|
99
72
|
raise ArgumentError, "Unknown compression type: #{type}"
|
|
100
73
|
end
|
|
101
74
|
end
|
|
102
75
|
|
|
103
|
-
def io_decompress_gzip(input, output)
|
|
104
|
-
loop do
|
|
105
|
-
reader = Zlib::GzipReader.new(input)
|
|
106
|
-
v = reader.read
|
|
107
|
-
output.write(v)
|
|
108
|
-
unused = reader.unused
|
|
109
|
-
reader.finish
|
|
110
|
-
unless unused.nil?
|
|
111
|
-
adjust = unused.length
|
|
112
|
-
input.pos -= adjust
|
|
113
|
-
end
|
|
114
|
-
break if input.eof?
|
|
115
|
-
end
|
|
116
|
-
output
|
|
117
|
-
end
|
|
118
|
-
|
|
119
|
-
def io_decompress_zstd(input, output)
|
|
120
|
-
reader = Zstd::StreamReader.new(input)
|
|
121
|
-
loop do
|
|
122
|
-
# Zstd::StreamReader needs to specify the size of the buffer
|
|
123
|
-
v = reader.read(1024)
|
|
124
|
-
output.write(v)
|
|
125
|
-
# Zstd::StreamReader doesn't provide unused data, so we have to manually adjust the position
|
|
126
|
-
break if input.eof?
|
|
127
|
-
end
|
|
128
|
-
output
|
|
129
|
-
end
|
|
130
|
-
|
|
131
76
|
def io_decompress(input, output, type = :gzip)
|
|
132
77
|
if type == :gzip
|
|
133
|
-
io_decompress_gzip(input, output)
|
|
78
|
+
Extractor.io_decompress_gzip(input, output, limit: @decompression_size_limit || DEFAULT_DECOMPRESSION_SIZE_LIMIT)
|
|
134
79
|
elsif type == :zstd
|
|
135
|
-
io_decompress_zstd(input, output)
|
|
80
|
+
Extractor.io_decompress_zstd(input, output, limit: @decompression_size_limit || DEFAULT_DECOMPRESSION_SIZE_LIMIT)
|
|
136
81
|
else
|
|
137
82
|
raise ArgumentError, "Unknown compression type: #{type}"
|
|
138
83
|
end
|