fluentd 1.19.2 → 1.19.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +81 -0
  3. data/Rakefile +7 -0
  4. data/lib/fluent/command/cat.rb +1 -1
  5. data/lib/fluent/config/basic_parser.rb +0 -1
  6. data/lib/fluent/config/literal_parser.rb +18 -7
  7. data/lib/fluent/config/types.rb +10 -2
  8. data/lib/fluent/configurable.rb +1 -0
  9. data/lib/fluent/daemon.rb +2 -1
  10. data/lib/fluent/engine.rb +1 -1
  11. data/lib/fluent/env.rb +1 -0
  12. data/lib/fluent/event.rb +2 -1
  13. data/lib/fluent/plugin/base.rb +7 -2
  14. data/lib/fluent/plugin/buf_file.rb +4 -4
  15. data/lib/fluent/plugin/buf_file_single.rb +2 -2
  16. data/lib/fluent/plugin/buf_memory.rb +1 -1
  17. data/lib/fluent/plugin/buffer/chunk.rb +16 -7
  18. data/lib/fluent/plugin/buffer/file_chunk.rb +7 -5
  19. data/lib/fluent/plugin/buffer/file_single_chunk.rb +7 -5
  20. data/lib/fluent/plugin/buffer/memory_chunk.rb +1 -1
  21. data/lib/fluent/plugin/buffer.rb +26 -5
  22. data/lib/fluent/plugin/compressable.rb +7 -62
  23. data/lib/fluent/plugin/extractor.rb +132 -0
  24. data/lib/fluent/plugin/filter_record_transformer.rb +2 -1
  25. data/lib/fluent/plugin/in_debug_agent.rb +1 -1
  26. data/lib/fluent/plugin/in_forward.rb +3 -1
  27. data/lib/fluent/plugin/in_http.rb +91 -6
  28. data/lib/fluent/plugin/in_monitor_agent.rb +21 -23
  29. data/lib/fluent/plugin/in_sample.rb +2 -1
  30. data/lib/fluent/plugin/in_syslog.rb +70 -4
  31. data/lib/fluent/plugin/out_file.rb +10 -0
  32. data/lib/fluent/plugin/out_forward/connection_manager.rb +18 -0
  33. data/lib/fluent/plugin/out_forward/socket_cache.rb +45 -10
  34. data/lib/fluent/plugin/out_forward.rb +13 -2
  35. data/lib/fluent/plugin/out_http.rb +31 -1
  36. data/lib/fluent/plugin/output.rb +69 -5
  37. data/lib/fluent/plugin/parser_csv.rb +5 -0
  38. data/lib/fluent/plugin/parser_json.rb +6 -1
  39. data/lib/fluent/plugin/parser_syslog.rb +3 -3
  40. data/lib/fluent/plugin/sd_file.rb +2 -1
  41. data/lib/fluent/plugin/storage_local.rb +4 -4
  42. data/lib/fluent/supervisor.rb +22 -1
  43. data/lib/fluent/test/base.rb +5 -0
  44. data/lib/fluent/version.rb +1 -1
  45. metadata +3 -3
  46. data/.deepsource.toml +0 -13
checksums.yaml CHANGED
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  SHA256:
3
- metadata.gz: 4873da89c533afd45a6f02a86aa8f92f7502bd08204724290dd34187c7183cd5
4
- data.tar.gz: 1f092049f456daae36077e374670a7a32d044da48b01ecea5410bc116f76c61b
3
+ metadata.gz: cab6a9d7ebeaec3be80f888c3b787fb8810fbd9bb78aad85aa5a96192d33365e
4
+ data.tar.gz: d80e36b87699060e8c895d9fb332dc521f0ba9cdbb0b0f4c80dd996003fdca50
5
5
  SHA512:
6
- metadata.gz: 820db817466aaade115300f5839e62d278a7fbbd19479b939206f4a0b212905046c1717c37cf2e19d90640dc07b570455e9b6f10fa5d21b04b96da4ec04f15b2
7
- data.tar.gz: 581e8ac6c84ae65e8b33758de56a65ca684547f64c2864fcdb630c68e2b58bda6232e98c60dc48b99efe8c2cf584f31aef62a14fd1a72785d1b6d88f41635fea
6
+ metadata.gz: 2ace1e9bde616fc42f958893bd9b8540f8047981f33bb7e22b8f595bf0296900125d8609976f45e2560f97b4cac40da5420adfa6d25589c0a04115093fb1aa40
7
+ data.tar.gz: 7af35f61b6d84663e4871f410e9f4d32116d59e5953ecf32797fbb3331f5caefff699d1a55820967b4b45d183ac0980a9776284e4d29ed94828264d5648d7992
data/CHANGELOG.md CHANGED
@@ -1,5 +1,86 @@
1
1
  # v1.19
2
2
 
3
+ ## Release v1.19.4 - 2026/09/28
4
+
5
+ ### Bug Fix
6
+
7
+ * buffer: enforce decompression_size_limit on the chunk IO path https://github.com/fluent/fluentd/pull/5504
8
+ * buffer: clamp exported buffer size metrics to non-negative values https://github.com/fluent/fluentd/pull/5488
9
+ * buffer: fix spurious `BufferOverflowError` caused by `queue_size` leaking when a chunk purge fails https://github.com/fluent/fluentd/pull/5487
10
+ * buffer: fix `stage_byte_size` leak on staged to unstaged chunk demotion that could eventually raise spurious `BufferOverflowError` in plugins implementing `#format` https://github.com/fluent/fluentd/pull/5456
11
+ * in_syslog: enforce message_length_limit on TCP/TLS transport https://github.com/fluent/fluentd/pull/5502
12
+ * The default value of `message_length_limit` is changed from 2048 to 8192 to match rsyslog's default `MaxMessageSize`.
13
+ * output: fix incomplete path traversal check in extract_placeholders https://github.com/fluent/fluentd/pull/5501
14
+ * output: treat JSON::GeneratorError as unrecoverable error https://github.com/fluent/fluentd/pull/5423
15
+ * parser_syslog: fix NameError when RFC5424 timestamp has repeated spaces https://github.com/fluent/fluentd/pull/5500
16
+ * parser_syslog: fix NameError when RFC3164 timestamp has repeated spaces https://github.com/fluent/fluentd/pull/5497
17
+ * parser_syslog: avoid excessive backtracking when parsing malformed RFC5424 structured data https://github.com/fluent/fluentd/pull/5444
18
+ * chunk: ensure to close the Tempfile for decompressed data https://github.com/fluent/fluentd/pull/5486
19
+ * plugin base: bound the number of worker lock files by hashing the path into a fixed set of buckets https://github.com/fluent/fluentd/pull/5471
20
+ * out_forward: stop the endless "ack in response and chunk id in sent data are different" warning storm by discarding (instead of reusing) a keepalive socket whose ack failed or came back with a mismatched chunk id https://github.com/fluent/fluentd/pull/5445
21
+ * config: accept empty lines in quoted strings https://github.com/fluent/fluentd/pull/5478
22
+ * config: fix a config error when a single scalar value is given to an array option in YAML config syntax (for example `retryable_response_codes: 503`) https://github.com/fluent/fluentd/pull/5433
23
+ * supervisor: reduce memory usage of cleanup_lock_dir with huge number of lock files https://github.com/fluent/fluentd/pull/5472
24
+
25
+ ### Enhancement
26
+
27
+ * in_http: add `<auth>` for basic authentication and `<security>` for client network allowlisting https://github.com/fluent/fluentd/pull/5503
28
+
29
+ ### Misc
30
+
31
+ * gem: support json gem v3.x https://github.com/fluent/fluentd/pull/5493
32
+ * Add `allow_comments: true` option for json parser https://github.com/fluent/fluentd/pull/5432
33
+ * Set `allow_duplicate_key: true` for all `JSON.parse` call https://github.com/fluent/fluentd/pull/5430
34
+ * CI fixes
35
+ * https://github.com/fluent/fluentd/pull/5495
36
+ * https://github.com/fluent/fluentd/pull/5477
37
+ * https://github.com/fluent/fluentd/pull/5466
38
+ * https://github.com/fluent/fluentd/pull/5460
39
+ * https://github.com/fluent/fluentd/pull/5455
40
+ * https://github.com/fluent/fluentd/pull/5454
41
+ * https://github.com/fluent/fluentd/pull/5453
42
+ * https://github.com/fluent/fluentd/pull/5452
43
+ * https://github.com/fluent/fluentd/pull/5434
44
+ * https://github.com/fluent/fluentd/pull/5426
45
+ * https://github.com/fluent/fluentd/pull/5425
46
+ * https://github.com/fluent/fluentd/pull/5424
47
+
48
+ ## Release v1.19.3 - 2026/06/25
49
+
50
+ ### Bug Fix
51
+
52
+ * out_http: add strict host validation for dynamic endpoints https://github.com/fluent/fluentd/pull/5394
53
+ * buffer, in_http: enforce size limits on decompressed payloads https://github.com/fluent/fluentd/pull/5393
54
+ * in_monitor_agent: change default visibility of config, retry, and debug info https://github.com/fluent/fluentd/pull/5392
55
+ * output: enforce strict path boundary validation for tag https://github.com/fluent/fluentd/pull/5391
56
+ * engine: remove duplicated word in unreloadable plugin error message https://github.com/fluent/fluentd/pull/5389
57
+ * storage_local: fix encoding error when reading non-ASCII characters https://github.com/fluent/fluentd/pull/5382
58
+ * parser_csv: skip empty or unparseable lines https://github.com/fluent/fluentd/pull/5359
59
+ * out_forward: avoid reusing closed keepalive sockets after remote disconnects https://github.com/fluent/fluentd/pull/5343
60
+ * buffer: resume buffer correctly even though path contains [] https://github.com/fluent/fluentd/pull/5305
61
+ * in_debug_agent: accept only from local machine by default https://github.com/fluent/fluentd/pull/5279
62
+
63
+ ### Misc
64
+
65
+ * gem: add win32-registry as runtime dependency for Ruby 4.1 https://github.com/fluent/fluentd/pull/5317
66
+ * output windows: check shorter service timeout on shutdown https://github.com/fluent/fluentd/pull/5306
67
+ * buffer: warn if default timekey (1d) will be used https://github.com/fluent/fluentd/pull/5291
68
+ * warn recommended exclusion path for antivirus https://github.com/fluent/fluentd/pull/5280
69
+ * CI fixes
70
+ * https://github.com/fluent/fluentd/pull/5387
71
+ * https://github.com/fluent/fluentd/pull/5386
72
+ * https://github.com/fluent/fluentd/pull/5366
73
+ * https://github.com/fluent/fluentd/pull/5342
74
+ * https://github.com/fluent/fluentd/pull/5341
75
+ * https://github.com/fluent/fluentd/pull/5340
76
+ * https://github.com/fluent/fluentd/pull/5339
77
+ * https://github.com/fluent/fluentd/pull/5338
78
+ * https://github.com/fluent/fluentd/pull/5337
79
+ * https://github.com/fluent/fluentd/pull/5336
80
+ * https://github.com/fluent/fluentd/pull/5335
81
+ * https://github.com/fluent/fluentd/pull/5334
82
+ * https://github.com/fluent/fluentd/pull/5333
83
+
3
84
  ## Release v1.19.2 - 2026/02/13
4
85
 
5
86
  ### Bug Fix
data/Rakefile CHANGED
@@ -78,4 +78,11 @@ task :coverity do
78
78
  FileUtils.rm_rf(['./cov-int', 'cov-fluentd.tar.gz'])
79
79
  end
80
80
 
81
+ task :check_env do
82
+ unless ENV['GEM_HOST_API_KEY']
83
+ abort "Missing required environment variable: GEM_HOST_API_KEY\nSee https://guides.rubygems.org/api-key-scopes/"
84
+ end
85
+ end
86
+ Rake::Task["release:rubygem_push"].enhance(["check_env"])
87
+
81
88
  task default: [:test, :build]
@@ -326,7 +326,7 @@ case format
326
326
  when 'json'
327
327
  begin
328
328
  while line = $stdin.gets
329
- record = Yajl.load(line)
329
+ record = JSON.parse(line, **Fluent::DEFAULT_JSON_PARSE_OPTIONS)
330
330
  w.write(record)
331
331
  end
332
332
  rescue
@@ -28,7 +28,6 @@ module Fluent
28
28
  SPACING = /(?:[ \t\r\n]|\z|\#.*?(?:\z|[\r\n]))+/
29
29
  ZERO_OR_MORE_SPACING = /(?:[ \t\r\n]|\z|\#.*?(?:\z|[\r\n]))*/
30
30
  SPACING_WITHOUT_COMMENT = /(?:[ \t\r\n]|\z)+/
31
- LINE_END_WITHOUT_SPACING_AND_COMMENT = /(?:\z|[\r\n])/
32
31
 
33
32
  module ClassMethods
34
33
  def symbol(string)
@@ -21,11 +21,21 @@ require 'yajl'
21
21
  require 'socket'
22
22
  require 'ripper'
23
23
 
24
+ require 'fluent/env'
24
25
  require 'fluent/config/basic_parser'
25
26
 
26
27
  module Fluent
27
28
  module Config
28
29
  class LiteralParser < BasicParser
30
+ # A physical line break (LF, CR, or CRLF) inside a quoted string.
31
+ # CRLF is normalized to LF so that the same config text does not produce a
32
+ # different value depending on whether the file was saved with LF or CRLF.
33
+ # A lone CR is kept as-is, and an escaped "\r\n" still produces CRLF.
34
+ LINE_BREAK = /\r\n|[\r\n]/
35
+ # A backslash immediately followed by a physical line break.
36
+ # It works as a line continuation, so both are stripped from the value.
37
+ LINE_CONTINUATION = /\\#{LINE_BREAK}/o
38
+
29
39
  def self.unescape_char(c)
30
40
  case c
31
41
  when '"'
@@ -98,11 +108,10 @@ module Fluent
98
108
  else
99
109
  return string.join
100
110
  end
101
- elsif check(/[^"]#{LINE_END_WITHOUT_SPACING_AND_COMMENT}/o)
102
- if s = check(/[^\\]#{LINE_END_WITHOUT_SPACING_AND_COMMENT}/o)
103
- string << s
104
- end
105
- skip(/[^"]#{LINE_END_WITHOUT_SPACING_AND_COMMENT}/o)
111
+ elsif skip(LINE_CONTINUATION)
112
+ next
113
+ elsif s = scan(LINE_BREAK)
114
+ string << (s == "\r\n" ? "\n" : s)
106
115
  elsif s = scan(/\\./)
107
116
  string << eval_escape_char(s[1,1])
108
117
  elsif skip(/\#\{/)
@@ -125,6 +134,8 @@ module Fluent
125
134
  string << "'"
126
135
  elsif s = scan(/\\\\/)
127
136
  string << "\\"
137
+ elsif s = scan(LINE_BREAK)
138
+ string << (s == "\r\n" ? "\n" : s)
128
139
  elsif s = scan(/./)
129
140
  string << s
130
141
  else
@@ -241,7 +252,7 @@ EOM
241
252
  # '{"foo":"bar", #' -> '{"foo":"bar"}' (to check)
242
253
  parsed = nil
243
254
  begin
244
- parsed = JSON.parse(buffer + line_buffer.rstrip.sub(/,$/, '') + (is_array ? "]" : "}"))
255
+ parsed = JSON.parse(buffer + line_buffer.rstrip.sub(/,$/, '') + (is_array ? "]" : "}"), **Fluent::DEFAULT_JSON_PARSE_OPTIONS)
245
256
  rescue JSON::ParserError
246
257
  # This '#' is in json string literals
247
258
  end
@@ -276,7 +287,7 @@ EOM
276
287
 
277
288
  line_buffer << char
278
289
  begin
279
- result = JSON.parse(buffer + line_buffer)
290
+ result = JSON.parse(buffer + line_buffer, **Fluent::DEFAULT_JSON_PARSE_OPTIONS)
280
291
  rescue JSON::ParserError
281
292
  # Incomplete json string yet
282
293
  end
@@ -201,7 +201,7 @@ module Fluent
201
201
  return nil if val.nil?
202
202
 
203
203
  param = if val.is_a?(String)
204
- val.start_with?('{') ? JSON.parse(val) : Hash[val.strip.split(/\s*,\s*/).map{|v| v.split(':', 2)}]
204
+ val.start_with?('{') ? JSON.parse(val, **Fluent::DEFAULT_JSON_PARSE_OPTIONS) : Hash[val.strip.split(/\s*,\s*/).map{|v| v.split(':', 2)}]
205
205
  else
206
206
  val
207
207
  end
@@ -228,7 +228,15 @@ module Fluent
228
228
  return nil if val.nil?
229
229
 
230
230
  param = if val.is_a?(String)
231
- val.start_with?('[') ? JSON.parse(val) : val.strip.split(/\s*,\s*/)
231
+ val.start_with?('[') ? JSON.parse(val, **Fluent::DEFAULT_JSON_PARSE_OPTIONS) : val.strip.split(/\s*,\s*/)
232
+ elsif val.is_a?(Array)
233
+ val
234
+ elsif val.is_a?(Numeric) || val == true || val == false
235
+ # Wrap only the bare scalars that Psych/YAML actually produces
236
+ # here (nil is handled by the early return above). Any other
237
+ # type (Hash, Symbol, Time, arbitrary objects) falls through and
238
+ # still raises "array required" below.
239
+ [val]
232
240
  else
233
241
  val
234
242
  end
@@ -14,6 +14,7 @@
14
14
  # limitations under the License.
15
15
  #
16
16
 
17
+ require 'fluent/env'
17
18
  require 'fluent/config/configure_proxy'
18
19
  require 'fluent/config/section'
19
20
  require 'fluent/config/error'
data/lib/fluent/daemon.rb CHANGED
@@ -5,9 +5,10 @@ here = File.dirname(__FILE__)
5
5
  $LOAD_PATH << File.expand_path(File.join(here, '..'))
6
6
 
7
7
  require 'serverengine'
8
+ require 'fluent/env'
8
9
  require 'fluent/supervisor'
9
10
 
10
11
  server_module = Fluent.const_get(ARGV[0])
11
12
  worker_module = Fluent.const_get(ARGV[1])
12
- params = JSON.parse(ARGV[2])
13
+ params = JSON.parse(ARGV[2], **Fluent::DEFAULT_JSON_PARSE_OPTIONS)
13
14
  ServerEngine::Daemon.run_server(server_module, worker_module) { Fluent::Supervisor.serverengine_config(params) }
data/lib/fluent/engine.rb CHANGED
@@ -186,7 +186,7 @@ module Fluent
186
186
 
187
187
  ret.all_plugins.each do |plugin|
188
188
  if plugin.respond_to?(:reloadable_plugin?) && !plugin.reloadable_plugin?
189
- raise Fluent::ConfigError, "Unreloadable plugin plugin: #{Fluent::Plugin.lookup_type_from_class(plugin.class)}, plugin_id: #{plugin.plugin_id}, class_name: #{plugin.class})"
189
+ raise Fluent::ConfigError, "Unreloadable plugin: #{Fluent::Plugin.lookup_type_from_class(plugin.class)}, plugin_id: #{plugin.plugin_id}, class_name: #{plugin.class})"
190
190
  end
191
191
  end
192
192
 
data/lib/fluent/env.rb CHANGED
@@ -26,6 +26,7 @@ module Fluent
26
26
  DEFAULT_SOCKET_PATH = ENV['FLUENT_SOCKET'] || '/var/run/fluent/fluent.sock'
27
27
  DEFAULT_BACKUP_DIR = ENV['FLUENT_BACKUP_DIR'] || '/tmp/fluent'
28
28
  DEFAULT_OJ_OPTIONS = Fluent::OjOptions.load_env
29
+ DEFAULT_JSON_PARSE_OPTIONS = { allow_duplicate_key: true, allow_comments: true }.freeze
29
30
  DEFAULT_DIR_PERMISSION = 0755
30
31
  DEFAULT_FILE_PERMISSION = 0644
31
32
  INSTANCE_ID = ENV['FLUENT_INSTANCE_ID'] || SecureRandom.uuid
data/lib/fluent/event.rb CHANGED
@@ -268,11 +268,12 @@ module Fluent
268
268
  end
269
269
 
270
270
  class CompressedMessagePackEventStream < MessagePackEventStream
271
- def initialize(data, cached_unpacker = nil, size = 0, unpacked_times: nil, unpacked_records: nil, compress: :gzip)
271
+ def initialize(data, cached_unpacker = nil, size = 0, unpacked_times: nil, unpacked_records: nil, compress: :gzip, decompression_size_limit: Fluent::Plugin::Compressable::DEFAULT_DECOMPRESSION_SIZE_LIMIT)
272
272
  super(data, cached_unpacker, size, unpacked_times: unpacked_times, unpacked_records: unpacked_records)
273
273
  @decompressed_data = nil
274
274
  @compressed_data = data
275
275
  @type = compress
276
+ @decompression_size_limit = decompression_size_limit
276
277
  end
277
278
 
278
279
  def empty?
@@ -14,6 +14,7 @@
14
14
  # limitations under the License.
15
15
  #
16
16
 
17
+ require 'zlib'
17
18
  require 'fluent/plugin'
18
19
  require 'fluent/configurable'
19
20
  require 'fluent/system_config'
@@ -69,9 +70,13 @@ module Fluent
69
70
  true
70
71
  end
71
72
 
73
+ LOCK_FILE_BUCKETS = 65536
74
+
72
75
  def get_lock_path(name)
73
- name = name.gsub(/[^a-zA-Z0-9]/, "_")
74
- File.join(@fluentd_lock_dir, "fluentd-#{name}.lock")
76
+ # The mapping from a name to a bucket MUST be identical across worker processes.
77
+ # Ruby's String#hash is randomly seeded per process and MUST NOT be used here.
78
+ bucket = Zlib.crc32(name.to_s) % LOCK_FILE_BUCKETS
79
+ File.join(@fluentd_lock_dir, "fluentd-bucket-#{bucket}.lock")
75
80
  end
76
81
 
77
82
  def acquire_worker_lock(name)
@@ -163,7 +163,7 @@ module Fluent
163
163
  end
164
164
 
165
165
  begin
166
- chunk = Fluent::Plugin::Buffer::FileChunk.new(m, path, mode, compress: @compress) # file chunk resumes contents of metadata
166
+ chunk = Fluent::Plugin::Buffer::FileChunk.new(m, path, mode, compress: @compress, decompression_size_limit: @decompression_size_limit) # file chunk resumes contents of metadata
167
167
  rescue Fluent::Plugin::Buffer::FileChunk::FileChunkError => e
168
168
  exist_broken_file = true
169
169
  handle_broken_files(path, mode, e)
@@ -205,7 +205,7 @@ module Fluent
205
205
  def generate_chunk(metadata)
206
206
  # FileChunk generates real path with unique_id
207
207
  perm = @file_permission || system_config.file_permission
208
- chunk = Fluent::Plugin::Buffer::FileChunk.new(metadata, @path, :create, perm: perm, compress: @compress)
208
+ chunk = Fluent::Plugin::Buffer::FileChunk.new(metadata, @path, :create, perm: perm, compress: @compress, decompression_size_limit: @decompression_size_limit)
209
209
  log.debug "Created new chunk", chunk_id: dump_unique_id_hex(chunk.unique_id), metadata: metadata
210
210
 
211
211
  return chunk
@@ -247,8 +247,8 @@ module Fluent
247
247
 
248
248
  def escaped_patterns(patterns)
249
249
  patterns.map { |pattern|
250
- # '{' '}' are special character in Dir.glob
251
- pattern.gsub(/[\{\}]/) { |c| "\\#{c}" }
250
+ # '{', '}', '[' and ']' are special character in Dir.glob
251
+ pattern.gsub(/[\{\}\[\]]/) { |c| "\\#{c}" }
252
252
  }
253
253
  end
254
254
  end
@@ -183,7 +183,7 @@ module Fluent
183
183
  end
184
184
 
185
185
  begin
186
- chunk = Fluent::Plugin::Buffer::FileSingleChunk.new(m, path, mode, @key_in_path, compress: @compress)
186
+ chunk = Fluent::Plugin::Buffer::FileSingleChunk.new(m, path, mode, @key_in_path, compress: @compress, decompression_size_limit: @decompression_size_limit)
187
187
  chunk.restore_size(@chunk_format) if @calc_num_records
188
188
  rescue Fluent::Plugin::Buffer::FileSingleChunk::FileChunkError => e
189
189
  exist_broken_file = true
@@ -216,7 +216,7 @@ module Fluent
216
216
  def generate_chunk(metadata)
217
217
  # FileChunk generates real path with unique_id
218
218
  perm = @file_permission || system_config.file_permission
219
- chunk = Fluent::Plugin::Buffer::FileSingleChunk.new(metadata, @path, :create, @key_in_path, perm: perm, compress: @compress)
219
+ chunk = Fluent::Plugin::Buffer::FileSingleChunk.new(metadata, @path, :create, @key_in_path, perm: perm, compress: @compress, decompression_size_limit: @decompression_size_limit)
220
220
 
221
221
  log.debug "Created new chunk", chunk_id: dump_unique_id_hex(chunk.unique_id), metadata: metadata
222
222
 
@@ -27,7 +27,7 @@ module Fluent
27
27
  end
28
28
 
29
29
  def generate_chunk(metadata)
30
- Fluent::Plugin::Buffer::MemoryChunk.new(metadata, compress: @compress)
30
+ Fluent::Plugin::Buffer::MemoryChunk.new(metadata, compress: @compress, decompression_size_limit: @decompression_size_limit)
31
31
  end
32
32
  end
33
33
  end
@@ -48,7 +48,7 @@ module Fluent
48
48
 
49
49
  # TODO: CompressedPackedMessage of forward protocol?
50
50
 
51
- def initialize(metadata, compress: :text)
51
+ def initialize(metadata, compress: :text, decompression_size_limit: Compressable::DEFAULT_DECOMPRESSION_SIZE_LIMIT)
52
52
  super()
53
53
  @unique_id = generate_unique_id
54
54
  @metadata = metadata
@@ -64,6 +64,7 @@ module Fluent
64
64
  elsif compress == :zstd
65
65
  extend ZstdDecompressable
66
66
  end
67
+ @decompression_size_limit = decompression_size_limit
67
68
  end
68
69
 
69
70
  attr_reader :unique_id, :metadata, :state
@@ -219,9 +220,13 @@ module Fluent
219
220
  Tempfile.new('decompressed-data')
220
221
  end
221
222
  output_io.binmode if output_io.is_a?(Tempfile)
222
- decompress(input_io: chunk_io, output_io: output_io)
223
- output_io.seek(0, IO::SEEK_SET)
224
- yield output_io
223
+ begin
224
+ decompress(input_io: chunk_io, output_io: output_io)
225
+ output_io.seek(0, IO::SEEK_SET)
226
+ yield output_io
227
+ ensure
228
+ output_io.close! if output_io.is_a?(Tempfile)
229
+ end
225
230
  end
226
231
  end
227
232
  end
@@ -273,9 +278,13 @@ module Fluent
273
278
  Tempfile.new('decompressed-data')
274
279
  end
275
280
  output_io.binmode if output_io.is_a?(Tempfile)
276
- decompress(input_io: chunk_io, output_io: output_io, type: :zstd)
277
- output_io.seek(0, IO::SEEK_SET)
278
- yield output_io
281
+ begin
282
+ decompress(input_io: chunk_io, output_io: output_io, type: :zstd)
283
+ output_io.seek(0, IO::SEEK_SET)
284
+ yield output_io
285
+ ensure
286
+ output_io.close! if output_io.is_a?(Tempfile)
287
+ end
279
288
  end
280
289
  end
281
290
  end
@@ -39,8 +39,8 @@ module Fluent
39
39
 
40
40
  attr_reader :path, :meta_path, :permission
41
41
 
42
- def initialize(metadata, path, mode, perm: nil, compress: :text)
43
- super(metadata, compress: compress)
42
+ def initialize(metadata, path, mode, perm: nil, compress: :text, decompression_size_limit: Compressable::DEFAULT_DECOMPRESSION_SIZE_LIMIT)
43
+ super(metadata, compress: compress, decompression_size_limit: decompression_size_limit)
44
44
  perm ||= Fluent::DEFAULT_FILE_PERMISSION
45
45
  @permission = perm.is_a?(String) ? perm.to_i(8) : perm
46
46
  @bytesize = @size = @adding_bytes = @adding_size = 0
@@ -168,9 +168,11 @@ module Fluent
168
168
 
169
169
  def open(**kwargs, &block)
170
170
  @chunk.seek(0, IO::SEEK_SET)
171
- val = yield @chunk
172
- @chunk.seek(0, IO::SEEK_END) if self.staged?
173
- val
171
+ begin
172
+ yield @chunk
173
+ ensure
174
+ @chunk.seek(0, IO::SEEK_END) if self.staged?
175
+ end
174
176
  end
175
177
 
176
178
  def self.assume_chunk_state(path)
@@ -35,8 +35,8 @@ module Fluent
35
35
 
36
36
  attr_reader :path, :permission
37
37
 
38
- def initialize(metadata, path, mode, key, perm: Fluent::DEFAULT_FILE_PERMISSION, compress: :text)
39
- super(metadata, compress: compress)
38
+ def initialize(metadata, path, mode, key, perm: Fluent::DEFAULT_FILE_PERMISSION, compress: :text, decompression_size_limit: Compressable::DEFAULT_DECOMPRESSION_SIZE_LIMIT)
39
+ super(metadata, compress: compress, decompression_size_limit: decompression_size_limit)
40
40
  @key = key
41
41
  perm ||= Fluent::DEFAULT_FILE_PERMISSION
42
42
  @permission = perm.is_a?(String) ? perm.to_i(8) : perm
@@ -141,9 +141,11 @@ module Fluent
141
141
 
142
142
  def open(**kwargs, &block)
143
143
  @chunk.seek(0, IO::SEEK_SET)
144
- val = yield @chunk
145
- @chunk.seek(0, IO::SEEK_END) if self.staged?
146
- val
144
+ begin
145
+ yield @chunk
146
+ ensure
147
+ @chunk.seek(0, IO::SEEK_END) if self.staged?
148
+ end
147
149
  end
148
150
 
149
151
  def self.assume_chunk_state(path)
@@ -20,7 +20,7 @@ module Fluent
20
20
  module Plugin
21
21
  class Buffer
22
22
  class MemoryChunk < Chunk
23
- def initialize(metadata, compress: :text)
23
+ def initialize(metadata, compress: :text, decompression_size_limit: Compressable::DEFAULT_DECOMPRESSION_SIZE_LIMIT)
24
24
  super
25
25
  @chunk = ''.force_encoding(Encoding::ASCII_8BIT)
26
26
  @chunk_bytes = 0
@@ -66,6 +66,9 @@ module Fluent
66
66
  desc 'Compress buffered data.'
67
67
  config_param :compress, :enum, list: [:text, :gzip, :zstd], default: :text
68
68
 
69
+ desc 'The size limit of the decompressed element.'
70
+ config_param :decompression_size_limit, :size, default: Compressable::DEFAULT_DECOMPRESSION_SIZE_LIMIT
71
+
69
72
  desc 'If true, chunks are thrown away when unrecoverable error happens'
70
73
  config_param :disable_chunk_backup, :bool, default: false
71
74
 
@@ -598,13 +601,19 @@ module Fluent
598
601
  metadata = chunk.metadata
599
602
  log.on_trace { log.trace "purging a chunk", instance: self.object_id, chunk_id: dump_unique_id_hex(chunk_id), metadata: metadata }
600
603
 
604
+ bytesize = chunk.bytesize
601
605
  begin
602
- bytesize = chunk.bytesize
603
606
  chunk.purge
604
- @queue_size_metrics.sub(bytesize)
605
607
  rescue => e
606
608
  log.error "failed to purge buffer chunk", chunk_id: dump_unique_id_hex(chunk_id), error_class: e.class, error: e
607
609
  log.error_backtrace
610
+ ensure
611
+ # Always release the queued byte counter, even when purge raises
612
+ # (e.g. unlink/close failure on the buffer path). Otherwise
613
+ # @queue_size is never decremented for a dequeued chunk that is
614
+ # gone from both @queue and @dequeued, so the leak ratchets toward
615
+ # total_limit_size and storable? becomes permanently false (#5468).
616
+ @queue_size_metrics.sub(bytesize)
608
617
  end
609
618
 
610
619
  @dequeued_num[chunk.metadata] -= 1
@@ -847,7 +856,12 @@ module Fluent
847
856
  # As already processed content is kept after rollback, then unstaged chunk should be queued.
848
857
  # After that, re-process current split again.
849
858
  # New chunk should be allocated, to do it, modify @stage and so on.
850
- synchronize { @stage.delete(modified_metadata) }
859
+ synchronize do
860
+ @stage.delete(modified_metadata)
861
+ # Subtract this chunk's already-counted staged bytes here;
862
+ # otherwise @stage_size leaks upward over the buffer's lifetime.
863
+ @stage_size_metrics.sub(original_bytesize) if chunk.staged?
864
+ end
851
865
  staged_chunk_used = false
852
866
  chunk.unstaged!
853
867
  break
@@ -909,8 +923,15 @@ module Fluent
909
923
  ]
910
924
 
911
925
  def statistics
912
- stage_size, queue_size = @stage_size_metrics.get, @queue_size_metrics.get
913
- buffer_space = 1.0 - ((stage_size + queue_size * 1.0) / @total_limit_size)
926
+ # Export-only clamp: internal gauges may go transiently negative during
927
+ # the deferred stage_size add vs enqueue_chunk sub race (#5303, #2712).
928
+ # Clamping the gauge store itself would turn that into a permanent
929
+ # over-count and break Buffer#storable? -- keep raw gauge semantics.
930
+ stage_size = [@stage_size_metrics.get, 0].max
931
+ queue_size = [@queue_size_metrics.get, 0].max
932
+ denom = @total_limit_size.to_f
933
+ # denom > 0 already excludes 0/0 NaN; stage/queue are floored above.
934
+ buffer_space = denom > 0.0 ? (1.0 - (stage_size + queue_size).to_f / denom).clamp(0.0, 1.0) : 0.0
914
935
  @stage_length_metrics.set(@stage.size)
915
936
  @queue_length_metrics.set(@queue.size)
916
937
  @available_buffer_space_ratios_metrics.set(buffer_space * 100)
@@ -14,6 +14,7 @@
14
14
  # limitations under the License.
15
15
  #
16
16
 
17
+ require 'fluent/plugin/extractor'
17
18
  require 'stringio'
18
19
  require 'zlib'
19
20
  require 'zstd-ruby'
@@ -21,6 +22,8 @@ require 'zstd-ruby'
21
22
  module Fluent
22
23
  module Plugin
23
24
  module Compressable
25
+ DEFAULT_DECOMPRESSION_SIZE_LIMIT = 256 * 1024 * 1024
26
+
24
27
  def compress(data, type: :gzip, **kwargs)
25
28
  output_io = kwargs[:output_io]
26
29
  io = output_io || StringIO.new
@@ -60,79 +63,21 @@ module Fluent
60
63
 
61
64
  private
62
65
 
63
- def string_decompress_gzip(compressed_data)
64
- io = StringIO.new(compressed_data)
65
- out = ''
66
- loop do
67
- reader = Zlib::GzipReader.new(io)
68
- out << reader.read
69
- unused = reader.unused
70
- reader.finish
71
- unless unused.nil?
72
- adjust = unused.length
73
- io.pos -= adjust
74
- end
75
- break if io.eof?
76
- end
77
- out
78
- end
79
-
80
- def string_decompress_zstd(compressed_data)
81
- io = StringIO.new(compressed_data)
82
- reader = Zstd::StreamReader.new(io)
83
- out = ''
84
- loop do
85
- # Zstd::StreamReader needs to specify the size of the buffer
86
- out << reader.read(1024)
87
- # Zstd::StreamReader doesn't provide unused data, so we have to manually adjust the position
88
- break if io.eof?
89
- end
90
- out
91
- end
92
-
93
66
  def string_decompress(compressed_data, type = :gzip)
94
67
  if type == :gzip
95
- string_decompress_gzip(compressed_data)
68
+ Extractor.decompress_gzip(compressed_data, limit: @decompression_size_limit || DEFAULT_DECOMPRESSION_SIZE_LIMIT)
96
69
  elsif type == :zstd
97
- string_decompress_zstd(compressed_data)
70
+ Extractor.decompress_zstd(compressed_data, limit: @decompression_size_limit || DEFAULT_DECOMPRESSION_SIZE_LIMIT)
98
71
  else
99
72
  raise ArgumentError, "Unknown compression type: #{type}"
100
73
  end
101
74
  end
102
75
 
103
- def io_decompress_gzip(input, output)
104
- loop do
105
- reader = Zlib::GzipReader.new(input)
106
- v = reader.read
107
- output.write(v)
108
- unused = reader.unused
109
- reader.finish
110
- unless unused.nil?
111
- adjust = unused.length
112
- input.pos -= adjust
113
- end
114
- break if input.eof?
115
- end
116
- output
117
- end
118
-
119
- def io_decompress_zstd(input, output)
120
- reader = Zstd::StreamReader.new(input)
121
- loop do
122
- # Zstd::StreamReader needs to specify the size of the buffer
123
- v = reader.read(1024)
124
- output.write(v)
125
- # Zstd::StreamReader doesn't provide unused data, so we have to manually adjust the position
126
- break if input.eof?
127
- end
128
- output
129
- end
130
-
131
76
  def io_decompress(input, output, type = :gzip)
132
77
  if type == :gzip
133
- io_decompress_gzip(input, output)
78
+ Extractor.io_decompress_gzip(input, output, limit: @decompression_size_limit || DEFAULT_DECOMPRESSION_SIZE_LIMIT)
134
79
  elsif type == :zstd
135
- io_decompress_zstd(input, output)
80
+ Extractor.io_decompress_zstd(input, output, limit: @decompression_size_limit || DEFAULT_DECOMPRESSION_SIZE_LIMIT)
136
81
  else
137
82
  raise ArgumentError, "Unknown compression type: #{type}"
138
83
  end