iostreams 1.11.0 → 2.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +14 -13
- data/Rakefile +52 -0
- data/docs/CLAUDE.md +9 -0
- data/docs/config.md +157 -0
- data/docs/copy_files.md +75 -0
- data/docs/extensions.md +111 -0
- data/docs/formats.md +188 -0
- data/docs/index.md +388 -0
- data/docs/path.md +652 -0
- data/docs/pgp.md +436 -0
- data/docs/streams.md +337 -0
- data/docs/tutorial.md +483 -0
- data/docs/upgrading.md +217 -0
- data/lib/io_streams/builder.rb +71 -11
- data/lib/io_streams/bzip2/reader.rb +25 -2
- data/lib/io_streams/bzip2/writer.rb +26 -2
- data/lib/io_streams/encode/reader.rb +6 -2
- data/lib/io_streams/encode/writer.rb +9 -5
- data/lib/io_streams/errors.rb +4 -0
- data/lib/io_streams/gzip/reader.rb +5 -1
- data/lib/io_streams/gzip/writer.rb +11 -2
- data/lib/io_streams/io_streams.rb +156 -20
- data/lib/io_streams/line/reader.rb +9 -4
- data/lib/io_streams/line/writer.rb +1 -1
- data/lib/io_streams/path.rb +117 -8
- data/lib/io_streams/paths/file.rb +57 -11
- data/lib/io_streams/paths/http.rb +123 -9
- data/lib/io_streams/paths/matcher.rb +3 -3
- data/lib/io_streams/paths/s3.rb +69 -18
- data/lib/io_streams/paths/sftp/net_ssh.rb +104 -0
- data/lib/io_streams/paths/sftp.rb +103 -64
- data/lib/io_streams/pgp/reader.rb +63 -10
- data/lib/io_streams/pgp/writer.rb +111 -30
- data/lib/io_streams/pgp.rb +256 -71
- data/lib/io_streams/reader.rb +14 -5
- data/lib/io_streams/record/reader.rb +75 -6
- data/lib/io_streams/record/writer.rb +3 -4
- data/lib/io_streams/row/reader.rb +1 -1
- data/lib/io_streams/row/writer.rb +1 -1
- data/lib/io_streams/stream.rb +48 -37
- data/lib/io_streams/symmetric_encryption/reader.rb +6 -2
- data/lib/io_streams/symmetric_encryption/writer.rb +8 -4
- data/lib/io_streams/tabular/header.rb +49 -10
- data/lib/io_streams/tabular/parser/array.rb +0 -10
- data/lib/io_streams/tabular/parser/base.rb +10 -0
- data/lib/io_streams/tabular/parser/csv.rb +9 -36
- data/lib/io_streams/tabular/parser/fixed.rb +8 -6
- data/lib/io_streams/tabular/parser/psv.rb +6 -14
- data/lib/io_streams/tabular.rb +5 -10
- data/lib/io_streams/utils.rb +34 -2
- data/lib/io_streams/version.rb +1 -1
- data/lib/io_streams/writer.rb +16 -7
- data/lib/io_streams/xlsx/reader.rb +6 -2
- data/lib/io_streams/zip/reader.rb +4 -0
- data/lib/io_streams/zip/writer.rb +26 -10
- data/lib/iostreams.rb +0 -1
- metadata +46 -112
- data/lib/io_streams/deprecated.rb +0 -216
- data/lib/io_streams/tabular/utility/csv_row.rb +0 -105
- data/test/builder_test.rb +0 -311
- data/test/bzip2_reader_test.rb +0 -27
- data/test/bzip2_writer_test.rb +0 -56
- data/test/deprecated_test.rb +0 -121
- data/test/encode_reader_test.rb +0 -51
- data/test/encode_writer_test.rb +0 -90
- data/test/files/embedded_lines_test.csv +0 -7
- data/test/files/multiple_files.zip +0 -0
- data/test/files/spreadsheet.xlsx +0 -0
- data/test/files/test.csv +0 -4
- data/test/files/test.json +0 -3
- data/test/files/test.psv +0 -4
- data/test/files/text file.txt +0 -3
- data/test/files/text.txt +0 -3
- data/test/files/text.txt.bz2 +0 -0
- data/test/files/text.txt.gz +0 -0
- data/test/files/text.txt.gz.zip +0 -0
- data/test/files/text.zip +0 -0
- data/test/files/text.zip.gz +0 -0
- data/test/files/unclosed_quote_large_test.csv +0 -1658
- data/test/files/unclosed_quote_test.csv +0 -4
- data/test/files/unclosed_quote_test2.csv +0 -3
- data/test/gzip_reader_test.rb +0 -27
- data/test/gzip_writer_test.rb +0 -52
- data/test/io_streams_test.rb +0 -132
- data/test/line_reader_test.rb +0 -325
- data/test/line_writer_test.rb +0 -59
- data/test/minimal_file_reader.rb +0 -25
- data/test/path_test.rb +0 -55
- data/test/paths/file_test.rb +0 -213
- data/test/paths/http_test.rb +0 -34
- data/test/paths/matcher_test.rb +0 -120
- data/test/paths/s3_test.rb +0 -220
- data/test/paths/sftp_test.rb +0 -106
- data/test/pgp_reader_test.rb +0 -46
- data/test/pgp_test.rb +0 -267
- data/test/pgp_writer_test.rb +0 -130
- data/test/record_reader_test.rb +0 -60
- data/test/record_writer_test.rb +0 -82
- data/test/row_reader_test.rb +0 -35
- data/test/row_writer_test.rb +0 -56
- data/test/stream_test.rb +0 -577
- data/test/tabular_test.rb +0 -338
- data/test/test_helper.rb +0 -40
- data/test/utils_test.rb +0 -20
- data/test/xlsx_reader_test.rb +0 -37
- data/test/zip_reader_test.rb +0 -53
- data/test/zip_writer_test.rb +0 -48
|
@@ -16,7 +16,7 @@ module IOStreams
|
|
|
16
16
|
|
|
17
17
|
# When reading from a file also add the line reader stream
|
|
18
18
|
def self.file(file_name, original_file_name: file_name, delimiter: $/, **args)
|
|
19
|
-
IOStreams::Line::Reader.file(file_name,
|
|
19
|
+
IOStreams::Line::Reader.file(file_name, delimiter: delimiter) do |io|
|
|
20
20
|
yield new(io, original_file_name: original_file_name, **args)
|
|
21
21
|
end
|
|
22
22
|
end
|
|
@@ -35,11 +35,10 @@ module IOStreams
|
|
|
35
35
|
# format_options: [Hash]
|
|
36
36
|
# Any specialized format specific options. For example, `:fixed` format requires the file definition.
|
|
37
37
|
#
|
|
38
|
-
# columns [Array<String>]
|
|
38
|
+
# columns [Array<String|Symbol>]
|
|
39
39
|
# The header columns when the file does not include a header row.
|
|
40
40
|
# Note:
|
|
41
|
-
#
|
|
42
|
-
# with MongoDB when it converts symbol keys to strings.
|
|
41
|
+
# Column names are converted to strings.
|
|
43
42
|
#
|
|
44
43
|
# allowed_columns [Array<String>]
|
|
45
44
|
# List of columns to allow.
|
|
@@ -57,6 +56,10 @@ module IOStreams
|
|
|
57
56
|
# #as_hash will skip these additional columns entirely as if they were not in the file at all.
|
|
58
57
|
# false:
|
|
59
58
|
# Raises Tabular::InvalidHeader when a column is supplied that is not in the whitelist.
|
|
59
|
+
#
|
|
60
|
+
# Note:
|
|
61
|
+
# * `allowed_columns`, `required_columns` and `skip_unknown` only apply to every input, including JSON records,
|
|
62
|
+
# supplied `columns` and `cleanse_header: false`, when `IOStreams.enforce_column_restrictions?` is true.
|
|
60
63
|
def initialize(line_reader, cleanse_header: true, original_file_name: nil, **args)
|
|
61
64
|
unless line_reader.respond_to?(:each)
|
|
62
65
|
raise(ArgumentError, "Stream must be a IOStreams::Line::Reader or implement #each")
|
|
@@ -65,18 +68,84 @@ module IOStreams
|
|
|
65
68
|
@tabular = IOStreams::Tabular.new(file_name: original_file_name, **args)
|
|
66
69
|
@line_reader = line_reader
|
|
67
70
|
@cleanse_header = cleanse_header
|
|
71
|
+
@warned = false
|
|
72
|
+
|
|
73
|
+
# Supplied columns take the place of a header row, so apply the allowed and required columns to them.
|
|
74
|
+
restrict_columns if restricted? && !@tabular.header?
|
|
68
75
|
end
|
|
69
76
|
|
|
70
77
|
def each
|
|
71
78
|
@line_reader.each do |line|
|
|
72
79
|
if @tabular.header?
|
|
73
80
|
@tabular.parse_header(line)
|
|
74
|
-
|
|
81
|
+
if @cleanse_header
|
|
82
|
+
cleanse_columns
|
|
83
|
+
elsif restricted?
|
|
84
|
+
restrict_columns
|
|
85
|
+
end
|
|
75
86
|
else
|
|
76
|
-
yield @tabular.record_parse(line)
|
|
87
|
+
yield restrict(@tabular.record_parse(line))
|
|
77
88
|
end
|
|
78
89
|
end
|
|
79
90
|
end
|
|
91
|
+
|
|
92
|
+
private
|
|
93
|
+
|
|
94
|
+
def restricted?
|
|
95
|
+
@tabular.header.restricted?
|
|
96
|
+
end
|
|
97
|
+
|
|
98
|
+
def cleanse_columns
|
|
99
|
+
@tabular.header.cleanse!(rename: @cleanse_header)
|
|
100
|
+
end
|
|
101
|
+
|
|
102
|
+
# Apply the allowed and required columns to supplied columns, or to a header row read with
|
|
103
|
+
# `cleanse_header: false`. Unless `IOStreams.enforce_column_restrictions?`, only warn when they would
|
|
104
|
+
# change the columns.
|
|
105
|
+
def restrict_columns
|
|
106
|
+
return cleanse_columns if IOStreams.enforce_column_restrictions?
|
|
107
|
+
return if @warned
|
|
108
|
+
|
|
109
|
+
header = @tabular.header
|
|
110
|
+
columns = header.columns
|
|
111
|
+
changed = changed_by_restriction? do
|
|
112
|
+
copy = IOStreams::Tabular::Header.new(
|
|
113
|
+
columns: columns,
|
|
114
|
+
allowed_columns: header.allowed_columns,
|
|
115
|
+
required_columns: header.required_columns,
|
|
116
|
+
skip_unknown: header.skip_unknown
|
|
117
|
+
)
|
|
118
|
+
copy.cleanse!(rename: @cleanse_header)
|
|
119
|
+
copy.columns != columns
|
|
120
|
+
end
|
|
121
|
+
warn_restriction if changed
|
|
122
|
+
end
|
|
123
|
+
|
|
124
|
+
# Formats such as JSON have no header row, so apply the allowed and required columns to each record's keys.
|
|
125
|
+
# Unless `IOStreams.enforce_column_restrictions?`, only warn when they would change the record.
|
|
126
|
+
def restrict(record)
|
|
127
|
+
return record unless record.is_a?(Hash) && restricted? && @tabular.header.columns.nil?
|
|
128
|
+
return @tabular.header.restrict_hash(record, rename: @cleanse_header) if IOStreams.enforce_column_restrictions?
|
|
129
|
+
return record if @warned
|
|
130
|
+
|
|
131
|
+
warn_restriction if changed_by_restriction? { @tabular.header.restrict_hash(record, rename: @cleanse_header) != record }
|
|
132
|
+
record
|
|
133
|
+
end
|
|
134
|
+
|
|
135
|
+
def changed_by_restriction?
|
|
136
|
+
yield
|
|
137
|
+
rescue IOStreams::Errors::InvalidHeader
|
|
138
|
+
true
|
|
139
|
+
end
|
|
140
|
+
|
|
141
|
+
# Warn once per reader, since the same columns usually apply to every record.
|
|
142
|
+
def warn_restriction
|
|
143
|
+
@warned = true
|
|
144
|
+
IOStreams.logger&.warn(
|
|
145
|
+
"allowed_columns and required_columns are not applied to this input, but would change the records read. " \
|
|
146
|
+
"In v3.0 they will apply to every input. Set `IOStreams.enforce_column_restrictions = true` to apply them now."
|
|
147
|
+
)
|
|
148
|
+
end
|
|
80
149
|
end
|
|
81
150
|
end
|
|
82
151
|
end
|
|
@@ -18,7 +18,7 @@ module IOStreams
|
|
|
18
18
|
|
|
19
19
|
# When writing to a file also add the line writer stream
|
|
20
20
|
def self.file(file_name, original_file_name: file_name, delimiter: $/, **args, &block)
|
|
21
|
-
IOStreams::Line::Writer.file(file_name,
|
|
21
|
+
IOStreams::Line::Writer.file(file_name, delimiter: delimiter) do |io|
|
|
22
22
|
yield new(io, original_file_name: original_file_name, **args, &block)
|
|
23
23
|
end
|
|
24
24
|
end
|
|
@@ -37,11 +37,10 @@ module IOStreams
|
|
|
37
37
|
# format_options: [Hash]
|
|
38
38
|
# Any specialized format specific options. For example, `:fixed` format requires the file definition.
|
|
39
39
|
#
|
|
40
|
-
# columns [Array<String>]
|
|
40
|
+
# columns [Array<String|Symbol>]
|
|
41
41
|
# The header columns when the file does not include a header row.
|
|
42
42
|
# Note:
|
|
43
|
-
#
|
|
44
|
-
# with MongoDB when it converts symbol keys to strings.
|
|
43
|
+
# Column names are converted to strings.
|
|
45
44
|
#
|
|
46
45
|
# allowed_columns [Array<String>]
|
|
47
46
|
# List of columns to allow.
|
|
@@ -14,7 +14,7 @@ module IOStreams
|
|
|
14
14
|
|
|
15
15
|
# When reading from a file also add the line reader stream
|
|
16
16
|
def self.file(file_name, original_file_name: file_name, delimiter: $/, **args)
|
|
17
|
-
IOStreams::Line::Reader.file(file_name,
|
|
17
|
+
IOStreams::Line::Reader.file(file_name, delimiter: delimiter) do |io|
|
|
18
18
|
yield new(io, original_file_name: original_file_name, **args)
|
|
19
19
|
end
|
|
20
20
|
end
|
|
@@ -21,7 +21,7 @@ module IOStreams
|
|
|
21
21
|
|
|
22
22
|
# When writing to a file also add the line writer stream
|
|
23
23
|
def self.file(file_name, original_file_name: file_name, delimiter: $/, **args, &block)
|
|
24
|
-
IOStreams::Line::Writer.file(file_name,
|
|
24
|
+
IOStreams::Line::Writer.file(file_name, delimiter: delimiter) do |io|
|
|
25
25
|
yield new(io, original_file_name: original_file_name, **args, &block)
|
|
26
26
|
end
|
|
27
27
|
end
|
data/lib/io_streams/stream.rb
CHANGED
|
@@ -18,8 +18,8 @@ module IOStreams
|
|
|
18
18
|
# Example:
|
|
19
19
|
#
|
|
20
20
|
# IOStreams.path("tempfile2527").stream(:zip).stream(:pgp, passphrase: "receiver_passphrase").read
|
|
21
|
-
def stream(stream, **
|
|
22
|
-
builder.stream(stream, **
|
|
21
|
+
def stream(stream, **)
|
|
22
|
+
builder.stream(stream, **)
|
|
23
23
|
self
|
|
24
24
|
end
|
|
25
25
|
|
|
@@ -33,15 +33,15 @@ module IOStreams
|
|
|
33
33
|
# IOStreams.path("keep_safe.enc").option(:pgp, passphrase: "receiver_passphrase").read
|
|
34
34
|
#
|
|
35
35
|
# IOStreams.path(output_file_name).option(:pgp, passphrase: "receiver_passphrase").read
|
|
36
|
-
def option(stream, **
|
|
37
|
-
builder.option(stream, **
|
|
36
|
+
def option(stream, **)
|
|
37
|
+
builder.option(stream, **)
|
|
38
38
|
self
|
|
39
39
|
end
|
|
40
40
|
|
|
41
41
|
# Adds the options for the specified stream as an option,
|
|
42
42
|
# but if streams have already been added it is instead added as a stream.
|
|
43
|
-
def option_or_stream(stream, **
|
|
44
|
-
builder.option_or_stream(stream, **
|
|
43
|
+
def option_or_stream(stream, **)
|
|
44
|
+
builder.option_or_stream(stream, **)
|
|
45
45
|
self
|
|
46
46
|
end
|
|
47
47
|
|
|
@@ -87,27 +87,34 @@ module IOStreams
|
|
|
87
87
|
# end
|
|
88
88
|
#
|
|
89
89
|
# Notes:
|
|
90
|
-
# -
|
|
91
|
-
# 1. The
|
|
92
|
-
#
|
|
90
|
+
# - Newlines embedded within quoted fields are kept within the same line when
|
|
91
|
+
# 1. The resolved tabular format quotes its fields (e.g. CSV, whether detected from a
|
|
92
|
+
# `.csv` file name or set explicitly via `.format(:csv)`)
|
|
93
|
+
# 2. Or the `embedded_within` argument is supplied (e.g. `embedded_within: '"'`)
|
|
94
|
+
# - Pass `embedded_within: nil` to disable quote-aware line joining for a quoted format.
|
|
93
95
|
def each(mode = :line, **args, &block)
|
|
94
96
|
raise(ArgumentError, "Invalid mode: #{mode.inspect}") if mode == :stream
|
|
95
97
|
|
|
96
|
-
#
|
|
98
|
+
# Deliberately not returning an Enumerator when no block is given.
|
|
99
|
+
# The stream pipeline manages resources via block scope: every stream is opened with an
|
|
100
|
+
# `ensure` that closes the file handle, reaps the gpg subprocess, deletes temp files, etc.
|
|
101
|
+
# A Fiber-backed Enumerator (e.g. `to_enum(__method__, mode, **args)`) would leave that block
|
|
102
|
+
# suspended; if the caller abandons a partially-consumed enumerator, none of the cleanup runs
|
|
103
|
+
# until GC collects the Fiber, leaking file descriptors, gpg processes, and temp files.
|
|
97
104
|
reader(mode, **args) { |stream| stream.each(&block) }
|
|
98
105
|
end
|
|
99
106
|
|
|
100
107
|
# Returns a Reader for reading a file / stream
|
|
101
|
-
def reader(mode = :stream, **args, &
|
|
108
|
+
def reader(mode = :stream, **args, &)
|
|
102
109
|
case mode
|
|
103
110
|
when :stream
|
|
104
|
-
stream_reader(&
|
|
111
|
+
stream_reader(&)
|
|
105
112
|
when :line
|
|
106
|
-
line_reader(**args, &
|
|
113
|
+
line_reader(**args, &)
|
|
107
114
|
when :array
|
|
108
|
-
row_reader(**args, &
|
|
115
|
+
row_reader(**args, &)
|
|
109
116
|
when :hash
|
|
110
|
-
record_reader(**args, &
|
|
117
|
+
record_reader(**args, &)
|
|
111
118
|
else
|
|
112
119
|
raise(ArgumentError, "Invalid mode: #{mode.inspect}")
|
|
113
120
|
end
|
|
@@ -124,16 +131,16 @@ module IOStreams
|
|
|
124
131
|
end
|
|
125
132
|
|
|
126
133
|
# Returns a Writer for writing to a file / stream
|
|
127
|
-
def writer(mode = :stream, **args, &
|
|
134
|
+
def writer(mode = :stream, **args, &)
|
|
128
135
|
case mode
|
|
129
136
|
when :stream
|
|
130
|
-
stream_writer(&
|
|
137
|
+
stream_writer(&)
|
|
131
138
|
when :line
|
|
132
|
-
line_writer(**args, &
|
|
139
|
+
line_writer(**args, &)
|
|
133
140
|
when :array
|
|
134
|
-
row_writer(**args, &
|
|
141
|
+
row_writer(**args, &)
|
|
135
142
|
when :hash
|
|
136
|
-
record_writer(**args, &
|
|
143
|
+
record_writer(**args, &)
|
|
137
144
|
else
|
|
138
145
|
raise(ArgumentError, "Invalid mode: #{mode.inspect}")
|
|
139
146
|
end
|
|
@@ -171,7 +178,7 @@ module IOStreams
|
|
|
171
178
|
# IOStreams.path("target_file.json").copy_from("source_file_name.csv.gz", convert: false)
|
|
172
179
|
#
|
|
173
180
|
# # Advanced copy with custom stream conversions on source and target.
|
|
174
|
-
# source = IOStreams.path("source_file").stream(encoding: "BINARY")
|
|
181
|
+
# source = IOStreams.path("source_file").stream(:encode, encoding: "BINARY")
|
|
175
182
|
# IOStreams.path("target_file.pgp").option(:pgp, passphrase: "hello").copy_from(source)
|
|
176
183
|
def copy_from(source, convert: true, mode: nil, **args)
|
|
177
184
|
if convert
|
|
@@ -322,23 +329,27 @@ module IOStreams
|
|
|
322
329
|
@builder ||= IOStreams::Builder.new
|
|
323
330
|
end
|
|
324
331
|
|
|
325
|
-
def stream_reader(&
|
|
326
|
-
builder.reader(io_stream, &
|
|
332
|
+
def stream_reader(&)
|
|
333
|
+
builder.reader(io_stream, &)
|
|
327
334
|
end
|
|
328
335
|
|
|
329
|
-
def line_reader(embedded_within:
|
|
330
|
-
|
|
336
|
+
def line_reader(embedded_within: :auto, **args)
|
|
337
|
+
# `:auto` defers the decision to the resolved tabular format (e.g. CSV quotes with `"`),
|
|
338
|
+
# while distinguishing "not supplied" from an explicit value such as `nil` (disable) or
|
|
339
|
+
# `'"'` (force). Centralizing this in the builder keeps all format-based decisions there.
|
|
340
|
+
embedded_within = builder.quote_character if embedded_within == :auto
|
|
331
341
|
|
|
332
342
|
stream_reader do |io|
|
|
333
|
-
yield IOStreams::Line::Reader.new(
|
|
334
|
-
|
|
335
|
-
|
|
336
|
-
|
|
343
|
+
yield IOStreams::Line::Reader.new(
|
|
344
|
+
io,
|
|
345
|
+
embedded_within: embedded_within,
|
|
346
|
+
**args
|
|
347
|
+
)
|
|
337
348
|
end
|
|
338
349
|
end
|
|
339
350
|
|
|
340
351
|
# Iterate over a file / stream returning each line as an array, one at a time.
|
|
341
|
-
def row_reader(delimiter: nil, embedded_within:
|
|
352
|
+
def row_reader(delimiter: nil, embedded_within: :auto, **args)
|
|
342
353
|
line_reader(delimiter: delimiter, embedded_within: embedded_within) do |io|
|
|
343
354
|
yield IOStreams::Row::Reader.new(
|
|
344
355
|
io,
|
|
@@ -351,7 +362,7 @@ module IOStreams
|
|
|
351
362
|
end
|
|
352
363
|
|
|
353
364
|
# Iterate over a file / stream returning each line as a hash, one at a time.
|
|
354
|
-
def record_reader(delimiter: nil, embedded_within:
|
|
365
|
+
def record_reader(delimiter: nil, embedded_within: :auto, **args)
|
|
355
366
|
line_reader(delimiter: delimiter, embedded_within: embedded_within) do |io|
|
|
356
367
|
yield IOStreams::Record::Reader.new(
|
|
357
368
|
io,
|
|
@@ -363,20 +374,20 @@ module IOStreams
|
|
|
363
374
|
end
|
|
364
375
|
end
|
|
365
376
|
|
|
366
|
-
def stream_writer(&
|
|
367
|
-
builder.writer(io_stream, &
|
|
377
|
+
def stream_writer(&)
|
|
378
|
+
builder.writer(io_stream, &)
|
|
368
379
|
end
|
|
369
380
|
|
|
370
381
|
def line_writer(**args, &block)
|
|
371
|
-
return block.call(io_stream) if io_stream
|
|
382
|
+
return block.call(io_stream) if io_stream.is_a?(IOStreams::Line::Writer)
|
|
372
383
|
|
|
373
384
|
writer do |io|
|
|
374
|
-
IOStreams::Line::Writer.stream(io,
|
|
385
|
+
IOStreams::Line::Writer.stream(io, **args, &block)
|
|
375
386
|
end
|
|
376
387
|
end
|
|
377
388
|
|
|
378
389
|
def row_writer(delimiter: $/, **args, &block)
|
|
379
|
-
return block.call(io_stream) if io_stream
|
|
390
|
+
return block.call(io_stream) if io_stream.is_a?(IOStreams::Row::Writer)
|
|
380
391
|
|
|
381
392
|
line_writer(delimiter: delimiter) do |io|
|
|
382
393
|
IOStreams::Row::Writer.stream(
|
|
@@ -391,7 +402,7 @@ module IOStreams
|
|
|
391
402
|
end
|
|
392
403
|
|
|
393
404
|
def record_writer(delimiter: $/, **args, &block)
|
|
394
|
-
return block.call(io_stream) if io_stream
|
|
405
|
+
return block.call(io_stream) if io_stream.is_a?(IOStreams::Record::Writer)
|
|
395
406
|
|
|
396
407
|
line_writer(delimiter: delimiter) do |io|
|
|
397
408
|
IOStreams::Record::Writer.stream(
|
|
@@ -1,11 +1,15 @@
|
|
|
1
1
|
module IOStreams
|
|
2
2
|
module SymmetricEncryption
|
|
3
3
|
class Reader < IOStreams::Reader
|
|
4
|
+
def self.option_names
|
|
5
|
+
%i[buffer_size version]
|
|
6
|
+
end
|
|
7
|
+
|
|
4
8
|
# read from a file/stream using Symmetric Encryption
|
|
5
|
-
def self.stream(input_stream, **args, &
|
|
9
|
+
def self.stream(input_stream, **args, &)
|
|
6
10
|
Utils.load_soft_dependency("symmetric-encryption", ".enc streaming") unless defined?(SymmetricEncryption)
|
|
7
11
|
|
|
8
|
-
::SymmetricEncryption::Reader.open(input_stream, **args, &
|
|
12
|
+
::SymmetricEncryption::Reader.open(input_stream, **args, &)
|
|
9
13
|
end
|
|
10
14
|
end
|
|
11
15
|
end
|
|
@@ -1,21 +1,25 @@
|
|
|
1
1
|
module IOStreams
|
|
2
2
|
module SymmetricEncryption
|
|
3
3
|
class Writer < IOStreams::Writer
|
|
4
|
+
def self.option_names
|
|
5
|
+
%i[compress version cipher_name header random_key random_iv]
|
|
6
|
+
end
|
|
7
|
+
|
|
4
8
|
# Write to stream using Symmetric Encryption
|
|
5
9
|
# By default the output stream is compressed.
|
|
6
10
|
# If the input_stream is already compressed consider setting compress: false.
|
|
7
|
-
def self.stream(input_stream, compress: true, **args, &
|
|
11
|
+
def self.stream(input_stream, compress: true, **args, &)
|
|
8
12
|
Utils.load_soft_dependency("symmetric-encryption", ".enc streaming") unless defined?(SymmetricEncryption)
|
|
9
13
|
|
|
10
|
-
::SymmetricEncryption::Writer.open(input_stream, compress: compress, **args, &
|
|
14
|
+
::SymmetricEncryption::Writer.open(input_stream, compress: compress, **args, &)
|
|
11
15
|
end
|
|
12
16
|
|
|
13
17
|
# Write to stream using Symmetric Encryption
|
|
14
18
|
# By default the output stream is compressed unless the file_name extension indicates the file is already compressed.
|
|
15
|
-
def self.file(file_name, compress: nil, **args, &
|
|
19
|
+
def self.file(file_name, compress: nil, **args, &)
|
|
16
20
|
Utils.load_soft_dependency("symmetric-encryption", ".enc streaming") unless defined?(SymmetricEncryption)
|
|
17
21
|
|
|
18
|
-
::SymmetricEncryption::Writer.open(file_name, compress: compress, **args, &
|
|
22
|
+
::SymmetricEncryption::Writer.open(file_name, compress: compress, **args, &)
|
|
19
23
|
end
|
|
20
24
|
end
|
|
21
25
|
end
|
|
@@ -5,16 +5,16 @@ module IOStreams
|
|
|
5
5
|
# Column names that begin with this prefix have been rejected and should be ignored.
|
|
6
6
|
IGNORE_PREFIX = "__rejected__".freeze
|
|
7
7
|
|
|
8
|
-
attr_accessor :
|
|
8
|
+
attr_accessor :allowed_columns, :required_columns, :skip_unknown
|
|
9
|
+
attr_reader :columns
|
|
9
10
|
|
|
10
11
|
# Header
|
|
11
12
|
#
|
|
12
13
|
# Parameters
|
|
13
|
-
# columns [Array<String>]
|
|
14
|
+
# columns [Array<String|Symbol>]
|
|
14
15
|
# Columns in this header.
|
|
15
16
|
# Note:
|
|
16
|
-
#
|
|
17
|
-
# with MongoDB when it converts symbol keys to strings.
|
|
17
|
+
# Column names are converted to strings.
|
|
18
18
|
#
|
|
19
19
|
# allowed_columns [Array<String>]
|
|
20
20
|
# List of columns to allow.
|
|
@@ -33,12 +33,17 @@ module IOStreams
|
|
|
33
33
|
# false:
|
|
34
34
|
# Raises Tabular::InvalidHeader when a column is supplied that is not in the whitelist.
|
|
35
35
|
def initialize(columns: nil, allowed_columns: nil, required_columns: nil, skip_unknown: true)
|
|
36
|
-
@columns = columns
|
|
36
|
+
@columns = stringify(columns)
|
|
37
37
|
@required_columns = required_columns
|
|
38
38
|
@allowed_columns = allowed_columns
|
|
39
39
|
@skip_unknown = skip_unknown
|
|
40
40
|
end
|
|
41
41
|
|
|
42
|
+
# Set the columns in this header, converting the column names to strings.
|
|
43
|
+
def columns=(columns)
|
|
44
|
+
@columns = stringify(columns)
|
|
45
|
+
end
|
|
46
|
+
|
|
42
47
|
# Returns [Array<String>] list columns that were ignored during cleansing.
|
|
43
48
|
#
|
|
44
49
|
# Each column is cleansed as follows:
|
|
@@ -47,16 +52,22 @@ module IOStreams
|
|
|
47
52
|
# - Spaces and '-' are converted to '_'.
|
|
48
53
|
# - All characters except for letters, digits, and '_' are stripped.
|
|
49
54
|
#
|
|
55
|
+
# Parameters:
|
|
56
|
+
# rename [true|false]
|
|
57
|
+
# Whether to cleanse the column names as described above.
|
|
58
|
+
# When false, the column names are compared to `allowed_columns` and `required_columns` as-is.
|
|
59
|
+
# Default: true
|
|
60
|
+
#
|
|
50
61
|
# Notes:
|
|
51
62
|
# * So that rejected columns can be identified in subsequent steps, they will be prefixed with `__rejected__`.
|
|
52
63
|
# For example, `Unknown Column` would be cleansed as `__rejected__Unknown Column`.
|
|
53
64
|
# * Raises Tabular::InvalidHeader when there are no rejected columns left after cleansing.
|
|
54
|
-
def cleanse!
|
|
65
|
+
def cleanse!(rename: true)
|
|
55
66
|
return [] if columns.nil? || columns.empty?
|
|
56
67
|
|
|
57
68
|
ignored_columns = []
|
|
58
69
|
self.columns = columns.collect do |column|
|
|
59
|
-
cleansed = cleanse_column(column)
|
|
70
|
+
cleansed = rename ? cleanse_column(column) : column
|
|
60
71
|
if allowed_columns.nil? || allowed_columns.include?(cleansed)
|
|
61
72
|
cleansed
|
|
62
73
|
else
|
|
@@ -107,6 +118,26 @@ module IOStreams
|
|
|
107
118
|
end
|
|
108
119
|
end
|
|
109
120
|
|
|
121
|
+
# Returns [true|false] whether `allowed_columns` or `required_columns` restrict the columns.
|
|
122
|
+
def restricted?
|
|
123
|
+
!allowed_columns.nil? || !required_columns.nil?
|
|
124
|
+
end
|
|
125
|
+
|
|
126
|
+
# Returns [Hash] the supplied hash after applying `allowed_columns`, `required_columns` and `skip_unknown`
|
|
127
|
+
# to its keys, as if its keys were the header row.
|
|
128
|
+
#
|
|
129
|
+
# Used for formats such as JSON where each record supplies its own keys instead of a header row.
|
|
130
|
+
def restrict_hash(hash, rename: true)
|
|
131
|
+
header = self.class.new(
|
|
132
|
+
columns: hash.keys,
|
|
133
|
+
allowed_columns: allowed_columns,
|
|
134
|
+
required_columns: required_columns,
|
|
135
|
+
skip_unknown: skip_unknown
|
|
136
|
+
)
|
|
137
|
+
header.cleanse!(rename: rename)
|
|
138
|
+
header.to_hash(hash.values)
|
|
139
|
+
end
|
|
140
|
+
|
|
110
141
|
def to_array(row, cleanse = true)
|
|
111
142
|
if row.is_a?(Hash) && columns
|
|
112
143
|
row = cleanse_hash(row) if cleanse
|
|
@@ -127,19 +158,23 @@ module IOStreams
|
|
|
127
158
|
|
|
128
159
|
def array_to_hash(row)
|
|
129
160
|
h = {}
|
|
130
|
-
columns.each_with_index
|
|
161
|
+
columns.each_with_index do |col, i|
|
|
162
|
+
h[col] = row[i] unless IOStreams::Utils.blank?(col) || col.start_with?(IGNORE_PREFIX)
|
|
163
|
+
end
|
|
131
164
|
h
|
|
132
165
|
end
|
|
133
166
|
|
|
134
167
|
# Perform cleansing on returned Hash keys during the narrowing process.
|
|
135
168
|
# For example, avoids issues with case etc.
|
|
136
169
|
def cleanse_hash(hash)
|
|
137
|
-
|
|
170
|
+
hash = hash.transform_keys(&:to_s) unless hash.keys.all?(String)
|
|
171
|
+
allowed = columns.reject { |column| column.start_with?(IGNORE_PREFIX) }
|
|
172
|
+
unmatched = allowed - hash.keys
|
|
138
173
|
unless unmatched.empty?
|
|
139
174
|
hash = hash.dup
|
|
140
175
|
unmatched.each { |name| hash[cleanse_column(name)] = hash.delete(name) }
|
|
141
176
|
end
|
|
142
|
-
hash.slice(*
|
|
177
|
+
hash.slice(*allowed)
|
|
143
178
|
end
|
|
144
179
|
|
|
145
180
|
def cleanse_column(name)
|
|
@@ -149,6 +184,10 @@ module IOStreams
|
|
|
149
184
|
cleansed.gsub!(/\W+/, "")
|
|
150
185
|
cleansed
|
|
151
186
|
end
|
|
187
|
+
|
|
188
|
+
def stringify(columns)
|
|
189
|
+
columns&.collect { |column| column&.to_s }
|
|
190
|
+
end
|
|
152
191
|
end
|
|
153
192
|
end
|
|
154
193
|
end
|
|
@@ -3,16 +3,6 @@ module IOStreams
|
|
|
3
3
|
class Tabular
|
|
4
4
|
module Parser
|
|
5
5
|
class Array < Base
|
|
6
|
-
# Returns [Array<String>] the header row.
|
|
7
|
-
# Returns nil if the row is blank.
|
|
8
|
-
def parse_header(row)
|
|
9
|
-
unless row.is_a?(::Array)
|
|
10
|
-
raise(IOStreams::Errors::InvalidHeader, "Format is :array. Invalid input header: #{row.class.name}")
|
|
11
|
-
end
|
|
12
|
-
|
|
13
|
-
row
|
|
14
|
-
end
|
|
15
|
-
|
|
16
6
|
# Returns Array
|
|
17
7
|
def parse(row)
|
|
18
8
|
raise(IOStreams::Errors::TypeMismatch, "Format is :array. Invalid input: #{row.class.name}") unless row.is_a?(::Array)
|
|
@@ -2,6 +2,16 @@ module IOStreams
|
|
|
2
2
|
class Tabular
|
|
3
3
|
module Parser
|
|
4
4
|
class Base
|
|
5
|
+
# Returns [String] the quote character within which field delimiters and embedded
|
|
6
|
+
# newlines may appear for this format, or [nil] when the format has no such quoting.
|
|
7
|
+
#
|
|
8
|
+
# Used by the line reader to avoid treating a newline as a line ending when it is
|
|
9
|
+
# embedded within a quoted field (e.g. CSV). Defined at the class level since it is a
|
|
10
|
+
# static property of the format, independent of any per-instance format options.
|
|
11
|
+
def self.quote_character
|
|
12
|
+
nil
|
|
13
|
+
end
|
|
14
|
+
|
|
5
15
|
# Returns [true|false] whether a header row is required for this format.
|
|
6
16
|
def requires_header?
|
|
7
17
|
true
|
|
@@ -3,24 +3,9 @@ module IOStreams
|
|
|
3
3
|
class Tabular
|
|
4
4
|
module Parser
|
|
5
5
|
class Csv < Base
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
def initialize
|
|
10
|
-
@csv_parser = Utility::CSVRow.new
|
|
11
|
-
end
|
|
12
|
-
end
|
|
13
|
-
|
|
14
|
-
# Returns [Array<String>] the header row.
|
|
15
|
-
# Returns nil if the row is blank.
|
|
16
|
-
def parse_header(row)
|
|
17
|
-
return row if row.is_a?(::Array)
|
|
18
|
-
|
|
19
|
-
unless row.is_a?(String)
|
|
20
|
-
raise(IOStreams::Errors::InvalidHeader, "Format is :csv. Invalid input header: #{row.class.name}")
|
|
21
|
-
end
|
|
22
|
-
|
|
23
|
-
parse_line(row)
|
|
6
|
+
# CSV fields may contain embedded delimiters and newlines when wrapped in double quotes.
|
|
7
|
+
def self.quote_character
|
|
8
|
+
'"'
|
|
24
9
|
end
|
|
25
10
|
|
|
26
11
|
# Returns [Array] the parsed CSV line
|
|
@@ -40,26 +25,14 @@ module IOStreams
|
|
|
40
25
|
|
|
41
26
|
private
|
|
42
27
|
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
# but at least it works on Ruby 2.6 and above.
|
|
46
|
-
def parse_line(line)
|
|
47
|
-
return if IOStreams::Utils.blank?(line)
|
|
28
|
+
def parse_line(line)
|
|
29
|
+
return if IOStreams::Utils.blank?(line)
|
|
48
30
|
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
def render_array(array)
|
|
53
|
-
CSV.generate_line(array, encoding: "UTF-8", row_sep: "")
|
|
54
|
-
end
|
|
55
|
-
else
|
|
56
|
-
def parse_line(line)
|
|
57
|
-
csv_parser.parse(line)
|
|
58
|
-
end
|
|
31
|
+
CSV.parse_line(line)
|
|
32
|
+
end
|
|
59
33
|
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
end
|
|
34
|
+
def render_array(array)
|
|
35
|
+
CSV.generate_line(array, encoding: "UTF-8", row_sep: "")
|
|
63
36
|
end
|
|
64
37
|
end
|
|
65
38
|
end
|