iostreams 1.11.0 → 2.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (108) hide show
  1. checksums.yaml +4 -4
  2. data/README.md +14 -13
  3. data/Rakefile +52 -0
  4. data/docs/CLAUDE.md +9 -0
  5. data/docs/config.md +157 -0
  6. data/docs/copy_files.md +75 -0
  7. data/docs/extensions.md +111 -0
  8. data/docs/formats.md +188 -0
  9. data/docs/index.md +388 -0
  10. data/docs/path.md +652 -0
  11. data/docs/pgp.md +436 -0
  12. data/docs/streams.md +337 -0
  13. data/docs/tutorial.md +483 -0
  14. data/docs/upgrading.md +217 -0
  15. data/lib/io_streams/builder.rb +71 -11
  16. data/lib/io_streams/bzip2/reader.rb +25 -2
  17. data/lib/io_streams/bzip2/writer.rb +26 -2
  18. data/lib/io_streams/encode/reader.rb +6 -2
  19. data/lib/io_streams/encode/writer.rb +9 -5
  20. data/lib/io_streams/errors.rb +4 -0
  21. data/lib/io_streams/gzip/reader.rb +5 -1
  22. data/lib/io_streams/gzip/writer.rb +11 -2
  23. data/lib/io_streams/io_streams.rb +156 -20
  24. data/lib/io_streams/line/reader.rb +9 -4
  25. data/lib/io_streams/line/writer.rb +1 -1
  26. data/lib/io_streams/path.rb +117 -8
  27. data/lib/io_streams/paths/file.rb +57 -11
  28. data/lib/io_streams/paths/http.rb +123 -9
  29. data/lib/io_streams/paths/matcher.rb +3 -3
  30. data/lib/io_streams/paths/s3.rb +69 -18
  31. data/lib/io_streams/paths/sftp/net_ssh.rb +104 -0
  32. data/lib/io_streams/paths/sftp.rb +103 -64
  33. data/lib/io_streams/pgp/reader.rb +63 -10
  34. data/lib/io_streams/pgp/writer.rb +111 -30
  35. data/lib/io_streams/pgp.rb +256 -71
  36. data/lib/io_streams/reader.rb +14 -5
  37. data/lib/io_streams/record/reader.rb +75 -6
  38. data/lib/io_streams/record/writer.rb +3 -4
  39. data/lib/io_streams/row/reader.rb +1 -1
  40. data/lib/io_streams/row/writer.rb +1 -1
  41. data/lib/io_streams/stream.rb +48 -37
  42. data/lib/io_streams/symmetric_encryption/reader.rb +6 -2
  43. data/lib/io_streams/symmetric_encryption/writer.rb +8 -4
  44. data/lib/io_streams/tabular/header.rb +49 -10
  45. data/lib/io_streams/tabular/parser/array.rb +0 -10
  46. data/lib/io_streams/tabular/parser/base.rb +10 -0
  47. data/lib/io_streams/tabular/parser/csv.rb +9 -36
  48. data/lib/io_streams/tabular/parser/fixed.rb +8 -6
  49. data/lib/io_streams/tabular/parser/psv.rb +6 -14
  50. data/lib/io_streams/tabular.rb +5 -10
  51. data/lib/io_streams/utils.rb +34 -2
  52. data/lib/io_streams/version.rb +1 -1
  53. data/lib/io_streams/writer.rb +16 -7
  54. data/lib/io_streams/xlsx/reader.rb +6 -2
  55. data/lib/io_streams/zip/reader.rb +4 -0
  56. data/lib/io_streams/zip/writer.rb +26 -10
  57. data/lib/iostreams.rb +0 -1
  58. metadata +46 -112
  59. data/lib/io_streams/deprecated.rb +0 -216
  60. data/lib/io_streams/tabular/utility/csv_row.rb +0 -105
  61. data/test/builder_test.rb +0 -311
  62. data/test/bzip2_reader_test.rb +0 -27
  63. data/test/bzip2_writer_test.rb +0 -56
  64. data/test/deprecated_test.rb +0 -121
  65. data/test/encode_reader_test.rb +0 -51
  66. data/test/encode_writer_test.rb +0 -90
  67. data/test/files/embedded_lines_test.csv +0 -7
  68. data/test/files/multiple_files.zip +0 -0
  69. data/test/files/spreadsheet.xlsx +0 -0
  70. data/test/files/test.csv +0 -4
  71. data/test/files/test.json +0 -3
  72. data/test/files/test.psv +0 -4
  73. data/test/files/text file.txt +0 -3
  74. data/test/files/text.txt +0 -3
  75. data/test/files/text.txt.bz2 +0 -0
  76. data/test/files/text.txt.gz +0 -0
  77. data/test/files/text.txt.gz.zip +0 -0
  78. data/test/files/text.zip +0 -0
  79. data/test/files/text.zip.gz +0 -0
  80. data/test/files/unclosed_quote_large_test.csv +0 -1658
  81. data/test/files/unclosed_quote_test.csv +0 -4
  82. data/test/files/unclosed_quote_test2.csv +0 -3
  83. data/test/gzip_reader_test.rb +0 -27
  84. data/test/gzip_writer_test.rb +0 -52
  85. data/test/io_streams_test.rb +0 -132
  86. data/test/line_reader_test.rb +0 -325
  87. data/test/line_writer_test.rb +0 -59
  88. data/test/minimal_file_reader.rb +0 -25
  89. data/test/path_test.rb +0 -55
  90. data/test/paths/file_test.rb +0 -213
  91. data/test/paths/http_test.rb +0 -34
  92. data/test/paths/matcher_test.rb +0 -120
  93. data/test/paths/s3_test.rb +0 -220
  94. data/test/paths/sftp_test.rb +0 -106
  95. data/test/pgp_reader_test.rb +0 -46
  96. data/test/pgp_test.rb +0 -267
  97. data/test/pgp_writer_test.rb +0 -130
  98. data/test/record_reader_test.rb +0 -60
  99. data/test/record_writer_test.rb +0 -82
  100. data/test/row_reader_test.rb +0 -35
  101. data/test/row_writer_test.rb +0 -56
  102. data/test/stream_test.rb +0 -577
  103. data/test/tabular_test.rb +0 -338
  104. data/test/test_helper.rb +0 -40
  105. data/test/utils_test.rb +0 -20
  106. data/test/xlsx_reader_test.rb +0 -37
  107. data/test/zip_reader_test.rb +0 -53
  108. data/test/zip_writer_test.rb +0 -48
@@ -50,13 +50,13 @@ module IOStreams
50
50
  self
51
51
  end
52
52
 
53
- def option_or_stream(stream, **options)
53
+ def option_or_stream(stream, **)
54
54
  if streams
55
- stream(stream, **options)
55
+ stream(stream, **)
56
56
  elsif file_name
57
- option(stream, **options)
57
+ option(stream, **)
58
58
  else
59
- stream(stream, **options)
59
+ stream(stream, **)
60
60
  end
61
61
  end
62
62
 
@@ -67,12 +67,12 @@ module IOStreams
67
67
  options[stream] if options
68
68
  end
69
69
 
70
- def reader(io_stream, &block)
71
- execute(:reader, pipeline, io_stream, &block)
70
+ def reader(io_stream, &)
71
+ execute(:reader, pipeline, io_stream, &)
72
72
  end
73
73
 
74
- def writer(io_stream, &block)
75
- execute(:writer, pipeline, io_stream, &block)
74
+ def writer(io_stream, &)
75
+ execute(:writer, pipeline, io_stream, &)
76
76
  end
77
77
 
78
78
  # Returns [Hash<Symbol:Hash>] the pipeline of streams
@@ -105,6 +105,18 @@ module IOStreams
105
105
  @format = format
106
106
  end
107
107
 
108
+ # Returns [String] the quote character within which field delimiters and newlines may be
109
+ # embedded for the current tabular format, or [nil] when the format has no such quoting,
110
+ # or when the format cannot be determined.
111
+ #
112
+ # Used by the line reader to avoid treating a newline as a line ending when it is embedded
113
+ # within a quoted field (e.g. CSV). Delegates to the format's parser, so the per-format
114
+ # knowledge lives with the parser. Driven entirely by `format`, so an explicitly set format
115
+ # (e.g. `.format(:psv)`) overrides any extension auto-detected from the `file_name`.
116
+ def quote_character
117
+ format && IOStreams::Tabular.parser_class(format).quote_character
118
+ end
119
+
108
120
  private
109
121
 
110
122
  def build_pipeline
@@ -120,7 +132,7 @@ module IOStreams
120
132
  end
121
133
 
122
134
  def class_for_stream(type, stream)
123
- ext = IOStreams.extensions[stream.nil? ? nil : stream.to_sym] ||
135
+ ext = IOStreams.extensions[stream&.to_sym] ||
124
136
  raise(ArgumentError, "Unknown Stream type: #{stream.inspect}")
125
137
  ext.send("#{type}_class") || raise(ArgumentError, "No #{type} registered for Stream type: #{stream.inspect}")
126
138
  end
@@ -146,14 +158,62 @@ module IOStreams
146
158
  block.call(io_stream)
147
159
  elsif pipeline.size == 1
148
160
  stream, opts = pipeline.first
149
- class_for_stream(type, stream).open(io_stream, **opts, &block)
161
+ open_stream(type, stream, io_stream, opts, &block)
150
162
  else
151
163
  # Daisy chain multiple streams together
152
164
  last = pipeline.keys.inject(block) do |inner, stream_sym|
153
- ->(io) { class_for_stream(type, stream_sym).open(io, **pipeline[stream_sym], &inner) }
165
+ ->(io) { open_stream(type, stream_sym, io, pipeline[stream_sym], &inner) }
154
166
  end
155
167
  last.call(io_stream)
156
168
  end
157
169
  end
170
+
171
+ def open_stream(type, stream, io_stream, opts, &)
172
+ klass = class_for_stream(type, stream)
173
+ validate_options(type, stream, klass, opts)
174
+ klass.open(io_stream, **opts, &)
175
+ end
176
+
177
+ # Options are strict: an option the stream does not accept raises instead of being ignored.
178
+ # One option hash is shared by the reader and the writer for a stream, so when the option
179
+ # is only valid in the other direction the message says so.
180
+ #
181
+ # Streams registered via `IOStreams.register_extension` need not inherit from `IOStreams::Reader`
182
+ # or `IOStreams::Writer`, so a class that does not declare `option_names` is not validated here.
183
+ def validate_options(type, stream, klass, opts)
184
+ accepted = option_names(klass)
185
+ return if accepted.nil?
186
+
187
+ unknown = opts.keys - accepted
188
+ return if unknown.empty?
189
+
190
+ other_type = type == :reader ? :writer : :reader
191
+ other_names = option_names(IOStreams.extensions[stream].send("#{other_type}_class")) || []
192
+ other_only = unknown & other_names
193
+ invalid = unknown - other_names
194
+ direction = type == :reader ? "reading" : "writing"
195
+
196
+ messages = []
197
+ if other_only.any?
198
+ messages << "#{list(other_only)} only #{other_only.size == 1 ? 'applies' : 'apply'} when " \
199
+ "#{type == :reader ? 'writing' : 'reading'} a #{stream.inspect} stream and cannot be used when " \
200
+ "#{direction}. Configure a separate path or stream without #{other_only.size == 1 ? 'it' : 'them'} " \
201
+ "for #{direction}."
202
+ end
203
+ if invalid.any?
204
+ valid = accepted.empty? ? "none" : list(accepted)
205
+ messages << "Unknown #{invalid.size == 1 ? 'option' : 'options'} #{list(invalid)} when #{direction} " \
206
+ "a #{stream.inspect} stream. Valid options: #{valid}."
207
+ end
208
+ raise(ArgumentError, messages.join(" "))
209
+ end
210
+
211
+ def option_names(klass)
212
+ klass.option_names if klass.respond_to?(:option_names)
213
+ end
214
+
215
+ def list(names)
216
+ names.map(&:inspect).join(", ")
217
+ end
158
218
  end
159
219
  end
@@ -1,12 +1,35 @@
1
1
  module IOStreams
2
2
  module Bzip2
3
3
  class Reader < IOStreams::Reader
4
+ OPTION_NAMES = %i[autoclose first_only small].freeze
5
+
6
+ # Not declared until v3.0, so that an unknown option logs a warning instead of raising `ArgumentError`.
7
+ def self.option_names
8
+ nil
9
+ end
10
+
4
11
  # Read from a Bzip2 stream, decompressing the contents as it is read
5
- def self.stream(input_stream, **args)
12
+ #
13
+ # Any other option is ignored and logs a warning. It will raise `ArgumentError` in v3.0.
14
+ #
15
+ # Parameters are passed through to `Bzip2::FFI::Reader`:
16
+ # autoclose: [true|false]
17
+ # Close the input stream when the reader is closed.
18
+ # Default: false
19
+ #
20
+ # first_only: [true|false]
21
+ # Only decompress the first of any consecutive bzip2 structures in the input.
22
+ # Default: false
23
+ #
24
+ # small: [true|false]
25
+ # Use an alternative decompression algorithm that uses less memory but is slower.
26
+ # Default: false
27
+ def self.stream(input_stream, autoclose: false, first_only: false, small: false, **unknown)
28
+ Utils.warn_unknown_options(unknown, :bz2, "reading", OPTION_NAMES)
6
29
  Utils.load_soft_dependency("bzip2-ffi", "Bzip2", "bzip2/ffi") unless defined?(::Bzip2::FFI)
7
30
 
8
31
  begin
9
- io = ::Bzip2::FFI::Reader.new(input_stream, args)
32
+ io = ::Bzip2::FFI::Reader.new(input_stream, autoclose: autoclose, first_only: first_only, small: small)
10
33
  yield io
11
34
  ensure
12
35
  io&.close
@@ -1,12 +1,36 @@
1
1
  module IOStreams
2
2
  module Bzip2
3
3
  class Writer < IOStreams::Writer
4
+ OPTION_NAMES = %i[autoclose block_size work_factor].freeze
5
+
6
+ # Not declared until v3.0, so that an unknown option logs a warning instead of raising `ArgumentError`.
7
+ def self.option_names
8
+ nil
9
+ end
10
+
4
11
  # Write to a stream, compressing with Bzip2
5
- def self.stream(input_stream, original_file_name: nil, **args)
12
+ #
13
+ # Any other option is ignored and logs a warning. It will raise `ArgumentError` in v3.0.
14
+ #
15
+ # Parameters are passed through to `Bzip2::FFI::Writer`:
16
+ # autoclose: [true|false]
17
+ # Close the output stream when the writer is closed.
18
+ # Default: false
19
+ #
20
+ # block_size: [Integer]
21
+ # Compression block size, from 1 (100k) to 9 (900k).
22
+ # Default: 9
23
+ #
24
+ # work_factor: [Integer]
25
+ # How much effort to spend on highly repetitive input before falling back
26
+ # to a slower algorithm, from 0 to 250. 0 uses the libbz2 default.
27
+ # Default: 0
28
+ def self.stream(input_stream, autoclose: false, block_size: nil, work_factor: nil, **unknown)
29
+ Utils.warn_unknown_options(unknown, :bz2, "writing", OPTION_NAMES)
6
30
  Utils.load_soft_dependency("bzip2-ffi", "Bzip2", "bzip2/ffi") unless defined?(::Bzip2::FFI)
7
31
 
8
32
  begin
9
- io = ::Bzip2::FFI::Writer.new(input_stream, args)
33
+ io = ::Bzip2::FFI::Writer.new(input_stream, autoclose: autoclose, block_size: block_size, work_factor: work_factor)
10
34
  yield io
11
35
  ensure
12
36
  io&.close
@@ -1,9 +1,13 @@
1
1
  module IOStreams
2
2
  module Encode
3
3
  class Reader < IOStreams::Reader
4
+ def self.option_names
5
+ %i[encoding cleaner replace]
6
+ end
7
+
4
8
  attr_reader :encoding, :cleaner
5
9
 
6
- NOT_PRINTABLE = Regexp.compile(/[^[:print:]|\r|\n]/).freeze
10
+ NOT_PRINTABLE = /[^[:print:]|\r\n]/
7
11
  # Builtin strip options to apply after encoding the read data.
8
12
  CLEANSE_RULES = {
9
13
  # Strips all non printable characters
@@ -13,7 +17,7 @@ module IOStreams
13
17
  }.freeze
14
18
 
15
19
  # Read a line at a time from a file or stream
16
- def self.stream(input_stream, original_file_name: nil, **args)
20
+ def self.stream(input_stream, **args)
17
21
  yield new(input_stream, **args)
18
22
  end
19
23
 
@@ -1,10 +1,14 @@
1
1
  module IOStreams
2
2
  module Encode
3
3
  class Writer < IOStreams::Writer
4
+ def self.option_names
5
+ %i[encoding cleaner replace]
6
+ end
7
+
4
8
  attr_reader :encoding, :cleaner
5
9
 
6
10
  # Write a line at a time to a file or stream
7
- def self.stream(input_stream, original_file_name: nil, **args)
11
+ def self.stream(input_stream, **args)
8
12
  yield new(input_stream, **args)
9
13
  end
10
14
 
@@ -46,7 +50,7 @@ module IOStreams
46
50
  # Write a line to the output stream
47
51
  #
48
52
  # Example:
49
- # IOStreams.writer('a.txt', encoding: 'UTF-8') do |stream|
53
+ # IOStreams.path('a.txt').option(:encode, encoding: 'UTF-8').writer do |stream|
50
54
  # stream << 'first line' << 'second line'
51
55
  # end
52
56
  def <<(record)
@@ -54,13 +58,13 @@ module IOStreams
54
58
  self
55
59
  end
56
60
 
57
- # Write a line to the output stream followed by the delimiter.
61
+ # Encode data and write it to the output stream.
58
62
  # Returns [Integer] the number of bytes written.
59
63
  #
60
64
  # Example:
61
- # IOStreams.writer('a.txt', encoding: 'UTF-8') do |stream|
65
+ # IOStreams.path('a.txt').option(:encode, encoding: 'UTF-8').writer do |stream|
62
66
  # count = stream.write('first line')
63
- # puts "Wrote #{count} bytes to the output file, including the delimiter"
67
+ # puts "Wrote #{count} bytes to the output file"
64
68
  # end
65
69
  def write(data)
66
70
  return 0 if data.nil?
@@ -18,6 +18,10 @@ module IOStreams
18
18
  class CommunicationsFailure < Error
19
19
  end
20
20
 
21
+ # When a path is not within any of the allowed paths, see `IOStreams.add_allowed_path`.
22
+ class AccessDenied < Error
23
+ end
24
+
21
25
  # When the specified delimiter is not found in the supplied stream / file
22
26
  class DelimiterNotFound < Error
23
27
  end
@@ -1,8 +1,12 @@
1
1
  module IOStreams
2
2
  module Gzip
3
3
  class Reader < IOStreams::Reader
4
+ def self.option_names
5
+ []
6
+ end
7
+
4
8
  # Read from a gzip stream, decompressing the contents as it is read
5
- def self.stream(input_stream, original_file_name: nil)
9
+ def self.stream(input_stream)
6
10
  io = ::Zlib::GzipReader.new(input_stream)
7
11
  yield io
8
12
  ensure
@@ -1,9 +1,18 @@
1
1
  module IOStreams
2
2
  module Gzip
3
3
  class Writer < IOStreams::Writer
4
+ def self.option_names
5
+ %i[level]
6
+ end
7
+
4
8
  # Write to a stream, compressing with GZip
5
- def self.stream(input_stream, original_file_name: nil, &block)
6
- io = ::Zlib::GzipWriter.new(input_stream)
9
+ #
10
+ # Parameters
11
+ # level: [Integer]
12
+ # Compression level, from 0 (no compression) to 9 (best compression).
13
+ # Default: Zlib::DEFAULT_COMPRESSION
14
+ def self.stream(input_stream, level: nil, &block)
15
+ io = ::Zlib::GzipWriter.new(input_stream, level)
7
16
  block.call(io)
8
17
  ensure
9
18
  io&.close
@@ -23,7 +23,7 @@ module IOStreams
23
23
  # # => "/usr/local/sample"
24
24
  #
25
25
  # IOStreams.path("s3://mybucket/path/file.xls")
26
- # # => #<IOStreams::S3::Path:0x00007fec66e3a288, @path="s3://mybucket/path/file.xls">
26
+ # # => #<IOStreams::Paths::S3:0x00007fec66e3a288 @path="s3://mybucket/path/file.xls">
27
27
  #
28
28
  # IOStreams.path("s3://mybucket/path/file.xls").to_s
29
29
  # # => "s3://mybucket/path/file.xls"
@@ -36,10 +36,9 @@ module IOStreams
36
36
  #
37
37
  # For Files
38
38
  # IOStreams.path('blah.zip').option(:encode, encoding: 'BINARY').each(:line) { |line| puts line }
39
- # IOStreams.path('blah.zip').option(:encode, encoding: 'UTF-8').each(:line).first
40
- # IOStreams.path('blah.zip').option(:encode, encoding: 'UTF-8').each(:hash).last
41
- # IOStreams.path('blah.zip').option(:encode, encoding: 'UTF-8').each(:hash).size
42
- # IOStreams.path('blah.zip').option(:encode, encoding: 'UTF-8').reader.size
39
+ # IOStreams.path('blah.zip').option(:encode, encoding: 'UTF-8').each(:line) { |line| puts line }
40
+ # IOStreams.path('blah.zip').option(:encode, encoding: 'UTF-8').each(:hash) { |hash| p hash }
41
+ # IOStreams.path('blah.zip').option(:encode, encoding: 'UTF-8').read
43
42
  # IOStreams.path('blah.csv.zip').each(:line) { |line| puts line }
44
43
  # IOStreams.path('blah.zip').option(:pgp, passphrase: 'receiver_passphrase').read
45
44
  # IOStreams.path('blah.zip').stream(:zip).stream(:pgp, passphrase: 'receiver_passphrase').read
@@ -75,14 +74,25 @@ module IOStreams
75
74
 
76
75
  # Join the supplied path elements to a root path.
77
76
  #
77
+ # Roots allow paths to reference a particular root directory, so that all path names
78
+ # are appended to that root. Use `IOStreams.join` instead of `IOStreams.path` so that
79
+ # the exact same code can run in production and development, yet use completely
80
+ # different data sources in each. For example, in production the root can point to
81
+ # an S3 bucket, while in development it points to the local file system.
82
+ #
83
+ # Roots are configured via an initializer at startup. Multiple roots can be setup,
84
+ # for example one for input files, another for output files, another for reports, etc.
85
+ # The `:default` root is used whenever a root is not supplied when calling `IOStreams.join`.
86
+ #
78
87
  # Example:
79
88
  # IOStreams.add_root(:default, "tmp/export")
89
+ # IOStreams.add_root(:ftp, "tmp/ftp")
80
90
  #
81
91
  # IOStreams.join('file.xls')
82
- # # => #<IOStreams::Paths::File:0x00007fec70391bd8 @path="tmp/export/sample">
92
+ # # => #<IOStreams::Paths::File:0x00007fec70391bd8 @path="tmp/export/file.xls">
83
93
  #
84
94
  # IOStreams.join('file.xls').to_s
85
- # # => "tmp/export/sample"
95
+ # # => "tmp/export/file.xls"
86
96
  #
87
97
  # IOStreams.join('sample', 'file.xls', root: :ftp)
88
98
  # # => #<IOStreams::Paths::File:0x00007fec6ee329b8 @path="tmp/ftp/sample/file.xls">
@@ -108,9 +118,13 @@ module IOStreams
108
118
  # Optional extension to add to the tempfile.
109
119
  #
110
120
  # Example:
111
- # IOStreams.temp_file
121
+ # IOStreams.temp_file("export", ".csv") { |path| path.write("Hello World") }
122
+ #
123
+ # Note: The temp file is accessible even when it is not within the allowed paths, see `IOStreams.add_allowed_path`.
112
124
  def self.temp_file(basename, extension = "")
113
- Utils.temp_file_name(basename, extension) { |file_name| yield(Paths::File.new(file_name).stream(:none)) }
125
+ Utils.temp_file_name(basename, extension) do |file_name|
126
+ yield(Paths::File.new(file_name).send(:permit!).stream(:none))
127
+ end
114
128
  end
115
129
 
116
130
  # Returns [IOStreams::Paths::File] current or named users home path
@@ -193,9 +207,9 @@ module IOStreams
193
207
  # "\a" "a" true # escaped ordinary remains ordinary
194
208
  # "[\?]" "?" true # can escape inside bracket expression
195
209
  #
196
- # "*" ".profile" false # wildcard doesn't match leading
197
- # "*" ".profile" true # period by default.
198
- # ".*" ".profile" true {hidden: true}
210
+ # "*" ".profile" false # wildcard doesn't match leading period by default
211
+ # "*" ".profile" true # unless hidden is enabled {hidden: true}
212
+ # ".*" ".profile" true # leading period is explicit
199
213
  #
200
214
  # "**/*.rb" "main.rb" false
201
215
  # "**/*.rb" "./main.rb" false
@@ -231,10 +245,81 @@ module IOStreams
231
245
  @root_paths.dup
232
246
  end
233
247
 
248
+ # Restrict IOStreams to only access paths within the supplied path.
249
+ #
250
+ # Once any allowed path has been added, reading, writing, listing, deleting or otherwise accessing
251
+ # a path that is not within one of the allowed paths raises `IOStreams::Errors::AccessDenied`.
252
+ # This prevents an untrusted file name, for example one supplied by a user, from accessing anything
253
+ # else that the process can access.
254
+ #
255
+ # Parameters: Same as `IOStreams.path`
256
+ #
257
+ # Returns [String] the normalized path that was added, against which paths are compared.
258
+ #
259
+ # Example:
260
+ # IOStreams.add_allowed_path("/var/data/uploads")
261
+ # IOStreams.add_allowed_path("s3://my-bucket/exports")
262
+ #
263
+ # IOStreams.path("/var/data/uploads/file.csv").read
264
+ # IOStreams.path("/etc/passwd").read
265
+ # # => IOStreams::Errors::AccessDenied
266
+ #
267
+ # Notes:
268
+ # * By default no allowed paths are added, and every path is accessible.
269
+ # * Add allowed paths in an initializer at startup, where they cannot be changed by untrusted input.
270
+ # * Paths are normalized before they are compared, so `..` cannot be used to leave an allowed path:
271
+ # * Local file names are resolved to their real path, following symbolic links.
272
+ # A relative path is resolved against the current working directory when it is added.
273
+ # * For S3 the bucket must match, and keys containing `.` or `..` segments are denied.
274
+ # * For SFTP and HTTP the host and port must match, and `.` and `..` segments are resolved.
275
+ # * `#each_child` skips children that are not within the allowed paths, for example a symbolic link
276
+ # to a file elsewhere.
277
+ # * Temp files created by `IOStreams.temp_file` are always accessible.
278
+ # * Paths from a scheme registered with `IOStreams.register_scheme` are denied, unless its path class
279
+ # implements the private method `#allowed_location`.
280
+ # * A local file could be replaced with a symbolic link after it is checked but before it is opened.
281
+ # Allowed paths do not prevent this, so do not allow paths where untrusted users can create files.
282
+ def self.add_allowed_path(*elements, **args)
283
+ location = allowed_location(path(*elements, **args))
284
+ @allowed_paths_mutex.synchronize { @allowed_paths = (@allowed_paths + [location]).uniq.freeze }
285
+ location
286
+ end
287
+
288
+ # Removes a path previously added with `IOStreams.add_allowed_path`.
289
+ #
290
+ # Returns [String] the normalized path that was removed.
291
+ def self.delete_allowed_path(*elements, **args)
292
+ location = allowed_location(path(*elements, **args))
293
+ @allowed_paths_mutex.synchronize { @allowed_paths = (@allowed_paths - [location]).freeze }
294
+ location
295
+ end
296
+
297
+ # Returns [Array<String>] the normalized allowed paths, see `IOStreams.add_allowed_path`.
298
+ def self.allowed_paths
299
+ @allowed_paths
300
+ end
301
+
302
+ # Returns [true|false] whether the supplied path can be accessed, see `IOStreams.add_allowed_path`.
303
+ #
304
+ # Always true when no allowed paths have been added.
305
+ #
306
+ # Parameters: Same as `IOStreams.path`
307
+ def self.allowed_path?(*elements, **args)
308
+ path(*elements, **args).send(:allowed?)
309
+ end
310
+
311
+ def self.allowed_location(path)
312
+ path.send(:allowed_location)
313
+ rescue Errors::AccessDenied => e
314
+ raise(ArgumentError, e.message)
315
+ end
316
+
317
+ private_class_method :allowed_location
318
+
234
319
  # Set the temporary path to use when creating local temp files.
235
320
  def self.temp_dir=(temp_dir)
236
321
  temp_dir = File.expand_path(temp_dir)
237
- FileUtils.mkdir_p(temp_dir) unless ::File.exist?(temp_dir)
322
+ FileUtils.mkdir_p(temp_dir)
238
323
 
239
324
  @temp_dir = temp_dir
240
325
  end
@@ -249,6 +334,54 @@ module IOStreams
249
334
 
250
335
  @temp_dir = nil
251
336
 
337
+ # Apply `allowed_columns`, `required_columns` and `skip_unknown` to every input when reading records.
338
+ #
339
+ # When false, they only apply to a header row read from the file, and only when `cleanse_header` is true.
340
+ # They are ignored for JSON and `:hash` input, when `columns:` is supplied, and with `cleanse_header: false`,
341
+ # and a warning is logged when applying them would change the result.
342
+ #
343
+ # When true, they also apply to the supplied `columns:`, to a header row read with `cleanse_header: false`,
344
+ # and to the keys of each JSON or `:hash` record. JSON keys are cleansed like a header row, unless
345
+ # `cleanse_header` is false, unknown keys are skipped or raise `IOStreams::Errors::InvalidHeader`,
346
+ # and a record missing a required column raises `IOStreams::Errors::InvalidHeader`.
347
+ #
348
+ # Since the format is usually inferred from the file name, set this to true when the allowed columns
349
+ # restrict what an uploaded file can set, so that renaming the file to `.json` cannot bypass them.
350
+ #
351
+ # Default: false. It will default to true in v3.0.
352
+ #
353
+ # Example:
354
+ # IOStreams.enforce_column_restrictions = true
355
+ def self.enforce_column_restrictions=(enforce)
356
+ raise(ArgumentError, "enforce_column_restrictions must be true or false") unless [true, false].include?(enforce)
357
+
358
+ @enforce_column_restrictions = enforce
359
+ end
360
+
361
+ # Returns [true|false] whether column restrictions apply to every input, see `IOStreams.enforce_column_restrictions=`.
362
+ def self.enforce_column_restrictions?
363
+ @enforce_column_restrictions
364
+ end
365
+
366
+ @enforce_column_restrictions = false
367
+
368
+ # Returns [Logger] the logger used by IOStreams for debug logging.
369
+ #
370
+ # When SemanticLogger is loaded a SemanticLogger instance is used by default,
371
+ # otherwise no logging is performed unless a logger is assigned via #logger=.
372
+ def self.logger
373
+ @logger
374
+ end
375
+
376
+ # Replace the logger used by IOStreams.
377
+ #
378
+ # Set to nil to disable logging.
379
+ def self.logger=(logger)
380
+ @logger = logger
381
+ end
382
+
383
+ @logger = (SemanticLogger[IOStreams] if defined?(SemanticLogger::Logger))
384
+
252
385
  # Register a file extension and the reader and writer streaming classes
253
386
  #
254
387
  # Example:
@@ -257,7 +390,7 @@ module IOStreams
257
390
  def self.register_extension(extension, reader_class, writer_class)
258
391
  raise(ArgumentError, "Invalid extension #{extension.inspect}") unless extension.nil? || extension.to_s =~ /\A\w+\Z/
259
392
 
260
- @extensions[extension.nil? ? nil : extension.to_sym] = Extension.new(reader_class, writer_class)
393
+ @extensions[extension&.to_sym] = Extension.new(reader_class, writer_class)
261
394
  end
262
395
 
263
396
  # De-Register a file extension
@@ -265,7 +398,7 @@ module IOStreams
265
398
  # Returns [Symbol] the extension removed, or nil if the extension was not registered
266
399
  #
267
400
  # Example:
268
- # register_extension(:xls)
401
+ # deregister_extension(:xls)
269
402
  def self.deregister_extension(extension)
270
403
  raise(ArgumentError, "Invalid extension #{extension.inspect}") unless extension.to_s =~ /\A\w+\Z/
271
404
 
@@ -277,15 +410,14 @@ module IOStreams
277
410
  @extensions.dup
278
411
  end
279
412
 
280
- # Register a file extension and the reader and writer streaming classes
413
+ # Register a URI scheme and the path class that handles it
281
414
  #
282
415
  # Example:
283
- # # MyXls::Reader and MyXls::Writer must implement .open
284
- # register_scheme(:xls, MyXls::Reader, MyXls::Writer)
416
+ # register_scheme(:gcs, MyGoogleCloudStoragePath)
285
417
  def self.register_scheme(scheme, klass)
286
418
  raise(ArgumentError, "Invalid scheme #{scheme.inspect}") unless scheme.nil? || scheme.to_s =~ /\A\w+\Z/
287
419
 
288
- @schemes[scheme.nil? ? nil : scheme.to_sym] = klass
420
+ @schemes[scheme&.to_sym] = klass
289
421
  end
290
422
 
291
423
  def self.schemes
@@ -293,7 +425,7 @@ module IOStreams
293
425
  end
294
426
 
295
427
  def self.scheme(scheme_name)
296
- @schemes[scheme_name.nil? ? nil : scheme_name.to_sym] || raise(ArgumentError, "Unknown Scheme type: #{scheme_name.inspect}")
428
+ @schemes[scheme_name&.to_sym] || raise(ArgumentError, "Unknown Scheme type: #{scheme_name.inspect}")
297
429
  end
298
430
 
299
431
  Extension = Struct.new(:reader_class, :writer_class)
@@ -301,6 +433,10 @@ module IOStreams
301
433
  # Hold root paths
302
434
  @root_paths = {}
303
435
 
436
+ # Hold allowed paths. Replaced rather than modified, so that it can be read without a lock.
437
+ @allowed_paths = [].freeze
438
+ @allowed_paths_mutex = Mutex.new
439
+
304
440
  # A registry to hold formats for processing files during upload or download
305
441
  @extensions = {}
306
442
  @schemes = {}
@@ -6,7 +6,7 @@ module IOStreams
6
6
  # Prevent denial of service when a delimiter is not found before this number * `buffer_size` characters are read.
7
7
  MAX_BLOCKS_MULTIPLIER = 100
8
8
 
9
- LINEFEED_REGEXP = Regexp.compile(/\r\n|\n|\r/).freeze
9
+ LINEFEED_REGEXP = /\r\n|\n|\r/
10
10
 
11
11
  # Read a line at a time from a stream
12
12
  def self.stream(input_stream, **args)
@@ -43,8 +43,9 @@ module IOStreams
43
43
  # For CSV files set `embedded_within: '"'`
44
44
  #
45
45
  # Note:
46
- # * When using a line reader and the file_name ends with ".csv" then embedded_within is automatically set to `"`
47
- def initialize(input_stream, delimiter: nil, buffer_size: 65_536, embedded_within: nil, original_file_name: nil)
46
+ # * When reached via `IOStreams::Stream`, `embedded_within` defaults to the quote character of the
47
+ # resolved tabular format (e.g. `"` for CSV). See `IOStreams::Builder#quote_character`.
48
+ def initialize(input_stream, delimiter: nil, buffer_size: 65_536, embedded_within: nil)
48
49
  super(input_stream)
49
50
 
50
51
  @embedded_within = embedded_within
@@ -93,7 +94,9 @@ module IOStreams
93
94
  line = _readline
94
95
  if line && @embedded_within
95
96
  initial_line_number = @line_number
96
- while line.count(@embedded_within).odd?
97
+ # Count the delimiters incrementally, since recounting the whole line each time is quadratic.
98
+ embedded_count = line.count(@embedded_within)
99
+ while embedded_count.odd?
97
100
  if eof? || line.length > @buffer_size * 10
98
101
  raise(Errors::MalformedDataError.new(
99
102
  "Unbalanced delimited field, delimiter: #{@embedded_within}",
@@ -101,6 +104,7 @@ module IOStreams
101
104
  ))
102
105
  end
103
106
  line << @delimiter
107
+ embedded_count += @delimiter.count(@embedded_within)
104
108
  next_line = _readline
105
109
  if next_line.nil?
106
110
  raise(Errors::MalformedDataError.new(
@@ -109,6 +113,7 @@ module IOStreams
109
113
  ))
110
114
  end
111
115
  line << next_line
116
+ embedded_count += next_line.count(@embedded_within)
112
117
  end
113
118
  end
114
119
  line
@@ -24,7 +24,7 @@ module IOStreams
24
24
  # Add the specified delimiter after every record when writing it
25
25
  # to the output stream
26
26
  # Default: OS Specific. Linux: "\n"
27
- def initialize(output_stream, delimiter: $/, original_file_name: nil)
27
+ def initialize(output_stream, delimiter: $/)
28
28
  super(output_stream)
29
29
  @delimiter = delimiter
30
30
  end