dsv 0.12.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +7 -0
- data/CHANGELOG +951 -0
- data/Gemfile +5 -0
- data/LICENSE +21 -0
- data/README.md +140 -0
- data/Rakefile +22 -0
- data/TODO +24 -0
- data/dsv.gemspec +43 -0
- data/lib/DSV/File.rb +35 -0
- data/lib/DSV/String.rb +19 -0
- data/lib/DSV/VERSION.rb +6 -0
- data/lib/dsv.rb +472 -0
- data/test/VERSION_test.rb +20 -0
- data/test/dsv_test.rb +702 -0
- data/test/gemspec_test.rb +42 -0
- data/test/helper.rb +9 -0
- metadata +97 -0
data/Gemfile
ADDED
data/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2006-2026 thoran
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
data/README.md
ADDED
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
# dsv
|
|
2
|
+
|
|
3
|
+
Delimiter-separated values for Ruby: CSV and its relatives, read and written with any delimiter on either side, in three small files.
|
|
4
|
+
|
|
5
|
+
This began in November 2006 as `csv2to`, became `CSVFile`, then `SimpleCSV`, and was a personal library for twenty years before it was a gem. The 2011 case for it against the CSV parsers of the day was that it was smaller, cleaner, faster, and gave keyed access to a row's columns. The first, second and fourth still hold; the third is measured, not claimed, and will be reported when it is.
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
## Installation
|
|
9
|
+
|
|
10
|
+
Add this line to your application's Gemfile:
|
|
11
|
+
|
|
12
|
+
```ruby
|
|
13
|
+
gem 'dsv'
|
|
14
|
+
```
|
|
15
|
+
|
|
16
|
+
or install it directly:
|
|
17
|
+
|
|
18
|
+
```shell
|
|
19
|
+
$ gem install dsv
|
|
20
|
+
```
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
## Usage
|
|
24
|
+
|
|
25
|
+
### Reading
|
|
26
|
+
|
|
27
|
+
A row is a plain Hash keyed by the header, so it merges, serialises and compares as a Hash does:
|
|
28
|
+
|
|
29
|
+
```ruby
|
|
30
|
+
require 'dsv'
|
|
31
|
+
|
|
32
|
+
DSV.read('people.csv', headers: true)
|
|
33
|
+
# => [{"name" => "Ada", "born" => "1815"}, {"name" => "Charles", "born" => "1791"}]
|
|
34
|
+
```
|
|
35
|
+
|
|
36
|
+
Without a header row the keys are positions:
|
|
37
|
+
|
|
38
|
+
```ruby
|
|
39
|
+
DSV.read("1,2,3\n4,5,6\n")
|
|
40
|
+
# => [{0 => "1", 1 => "2", 2 => "3"}, {0 => "4", 1 => "5", 2 => "6"}]
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
A String that names an existing file is read as a file; any other String is read as text.
|
|
44
|
+
|
|
45
|
+
### Selecting columns
|
|
46
|
+
|
|
47
|
+
Name the columns wanted, by name or by position, and the rows hold only those:
|
|
48
|
+
|
|
49
|
+
```ruby
|
|
50
|
+
DSV.read('people.csv', 'name', headers: true)
|
|
51
|
+
# => [{"name" => "Ada"}, {"name" => "Charles"}]
|
|
52
|
+
|
|
53
|
+
DSV.read('people.csv', headers: true, selected_columns: ['name'])
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
### Enumerating from the class
|
|
57
|
+
|
|
58
|
+
The Enumerable calls are available on a source directly, each taking the same options:
|
|
59
|
+
|
|
60
|
+
```ruby
|
|
61
|
+
DSV.each('people.csv', headers: true){|row| puts row['name']}
|
|
62
|
+
DSV.select('people.csv', headers: true){|row| row['born'] < '1800'}
|
|
63
|
+
DSV.detect('people.csv', headers: true){|row| row['name'] == 'Ada'}
|
|
64
|
+
DSV.collect('people.csv', headers: true){|row| row['name'].upcase}
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
`foreach`, `map`, `find_all` and `find` are aliases.
|
|
68
|
+
|
|
69
|
+
### Any delimiter
|
|
70
|
+
|
|
71
|
+
The column separator is a String or a Regexp; the row separator is a String:
|
|
72
|
+
|
|
73
|
+
```ruby
|
|
74
|
+
DSV.read('people.tsv', headers: true, column_separator: "\t")
|
|
75
|
+
DSV.read('people.txt', headers: true, column_separator: /,\s*/)
|
|
76
|
+
DSV.read('people.csv', headers: true, row_separator: "\r\n")
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
`col_sep:` and `row_sep:` are accepted as spellings of those. Writing takes the same two options, so a file round-trips through whatever it was read with.
|
|
80
|
+
|
|
81
|
+
### Quoting
|
|
82
|
+
|
|
83
|
+
Unspecified, quoting follows RFC 4180: a quoted field may hold the separator or the row separator, a doubled quote inside it reads as one quote, and a row's quotes are read only when the row holds one. Naming a mode is faster, since each is a direct string operation on the contract it names:
|
|
84
|
+
|
|
85
|
+
```ruby
|
|
86
|
+
DSV.read('plain.csv', headers: true, quote: :none) # no field is quoted; quotes are data
|
|
87
|
+
DSV.read('quoted.csv', headers: true, quote: :double) # every field is quoted, as DSV writes them
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
### Writing
|
|
91
|
+
|
|
92
|
+
Set the columns, give the rows as Hashes, and write:
|
|
93
|
+
|
|
94
|
+
```ruby
|
|
95
|
+
DSV.open('out.csv', mode: 'w', headers: true, columns: ['name', 'born']) do |dsv|
|
|
96
|
+
dsv.rows = [{'name' => 'Ada', 'born' => 1815}]
|
|
97
|
+
dsv.write
|
|
98
|
+
end
|
|
99
|
+
# out.csv:
|
|
100
|
+
# "name","born"
|
|
101
|
+
# "Ada","1815"
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
Every field is quoted, which is the `:double` contract on the way back in, and a quote inside a value is doubled. `quote: :none` writes bare values, `:single` uses single quotes. A `nil` writes as an empty field. Rows without columns defined are written from their own keys, and if the keys are names they become the header.
|
|
105
|
+
|
|
106
|
+
Under `r+` a file is read, changed and written back in place; reading never alters the file, and the first write replaces it:
|
|
107
|
+
|
|
108
|
+
```ruby
|
|
109
|
+
DSV.open('people.csv', mode: 'r+', headers: true) do |dsv|
|
|
110
|
+
dsv.rows = dsv.read.collect{|row| row.merge('born' => row['born'].to_i + 1)}
|
|
111
|
+
dsv.write
|
|
112
|
+
end
|
|
113
|
+
```
|
|
114
|
+
|
|
115
|
+
### Repeated header names
|
|
116
|
+
|
|
117
|
+
Where the header repeats a name, `columns` maps the name to every position and the row holds the values under that name as an Array, in position order:
|
|
118
|
+
|
|
119
|
+
```ruby
|
|
120
|
+
DSV.columns("a,a,b\n1,2,3\n", headers: true) # => {"a" => [0, 1], "b" => 2}
|
|
121
|
+
DSV.read("a,a,b\n1,2,3\n", headers: true) # => [{"a" => ["1", "2"], "b" => "3"}]
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
So a value is a String, or an Array where the header repeats the name. `Array(row['a'])` reads a column that may be either, and `repeated_names` says which names those are. Writing spreads an Array back over its positions.
|
|
125
|
+
|
|
126
|
+
### One line
|
|
127
|
+
|
|
128
|
+
```ruby
|
|
129
|
+
DSV.parse_line("1,\"2,3\",4\n") # => ["1", "2,3", "4"]
|
|
130
|
+
```
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
## What it is not
|
|
134
|
+
|
|
135
|
+
There are no converters, no `Row` or `Table` classes, and no encoding handling: a value is always a String, a row is always a Hash. Anything wanted beyond that is in `TODO`, in the order it is likely to arrive.
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
## License
|
|
139
|
+
|
|
140
|
+
MIT. See `LICENSE`.
|
data/Rakefile
ADDED
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
# Rakefile
|
|
2
|
+
|
|
3
|
+
require 'rake/testtask'
|
|
4
|
+
|
|
5
|
+
Rake::TestTask.new(:test) do |t|
|
|
6
|
+
t.libs << 'lib'
|
|
7
|
+
t.libs << 'test'
|
|
8
|
+
t.test_files = FileList['test/**/*_test.rb']
|
|
9
|
+
t.verbose = true
|
|
10
|
+
t.warning = false
|
|
11
|
+
end
|
|
12
|
+
|
|
13
|
+
task default: :test
|
|
14
|
+
|
|
15
|
+
desc "Run tests"
|
|
16
|
+
task :spec => :test
|
|
17
|
+
|
|
18
|
+
desc "Show version"
|
|
19
|
+
task :version do
|
|
20
|
+
require_relative './lib/DSV/VERSION'
|
|
21
|
+
puts "dsv #{DSV::VERSION}"
|
|
22
|
+
end
|
data/TODO
ADDED
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
# TODO
|
|
2
|
+
|
|
3
|
+
## 0.12.x, before the gem is published
|
|
4
|
+
|
|
5
|
+
1. A path is told from content by the call, not by a guess: .read, .foreach, .open and the other path-taking class methods take a path, .parse takes text, new keeps the guess with a switch to say which, and a path under a read mode that names no file raises rather than being read as text.
|
|
6
|
+
2. Selection as a keyword on the calls, only: or select:, columns: being taken; the positional form kept or retired; selected_columns: renamed to match.
|
|
7
|
+
3. The instance-level header_row accessor renamed headers?, the class-level header_row now being the row.
|
|
8
|
+
4. << to append a row and generate to build a string, the two writing calls a CSV user reaches for first.
|
|
9
|
+
|
|
10
|
+
## 0.13.0, speed
|
|
11
|
+
|
|
12
|
+
5. The saved-up speed change, with the evidence in the reconciliation: two string predicates in place of four regex matches per piece in split_csv; rows built by zip.to_h; a benchmark script in the repository against stdlib CSV, and the README's speed sentence written from its output. The scanner-based parser on the branch scanner-parser is judged here, for the Regexp separator with a quoted field if for nothing else.
|
|
13
|
+
|
|
14
|
+
## Features, after
|
|
15
|
+
|
|
16
|
+
6. Guessing, as a mode that runs once at construction on a sample of the first rows and feeds the same chooser as an explicit option, scanner only: the column separator, the one whose count is the same on every sampled row; whether the first row is a header, its columns' types differing from the rows beneath in the same way across columns; and, where there is no header, a header made from the data, each column named for what its values look like, integer, decimal, date, email, URL, phone, with a positional fallback, column_0, for text and a suffix where two columns share a class. Names come from what values look like, not what they mean; anything more is a schema, and another library's.
|
|
17
|
+
7. Templated ingestion: a template: option taking a whole-row Regexp with a capture per field, the group names being the column names, one match per row; an Array of separators as its writable special case, compiled into such a Regexp at construction and invertible for writing where a general template is not.
|
|
18
|
+
8. The duplicate-header modes as an option: keep all as an Array, the default; index, a, a.1; first; last; error. All derivations of the same columns map at header time, none costing anything per row.
|
|
19
|
+
9. Writing options: minimal quoting, a field quoted only where it holds a separator, a quote or a row terminator; a row terminator inside a field kept, quoted, or folded to a space; the encoding declared and a transcode offered with any loss marked.
|
|
20
|
+
10. Symbols and Strings interchangeable for column names, in a selection, a columns: list and a row's keys; today a String is required.
|
|
21
|
+
11. A strict: option, off by default, that raises on a row with more or fewer fields than the header; today a short row gets only the columns it has and a long row's excess lands under nil.
|
|
22
|
+
12. A line number in each parse error, UnterminatedQuote first, found only when the error is raised, from the source's position at that moment; never counted as it goes, since counting costs time on every row for the sake of the rare one.
|
|
23
|
+
13. Chunked processing of a large file: a bounded number of rows at a time handed to a block, so that nothing needs the whole file in memory and a partial result can be acted on before the rest is read.
|
|
24
|
+
14. Checks, all of them after the above: strict: (13), column types, and whatever else proves wanted; correct and published comes first.
|
data/dsv.gemspec
ADDED
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
# dsv.gemspec
|
|
2
|
+
|
|
3
|
+
require_relative './lib/DSV/VERSION'
|
|
4
|
+
|
|
5
|
+
class Gem::Specification
|
|
6
|
+
def development_dependencies=(gems)
|
|
7
|
+
gems.each{|gem| add_development_dependency(*gem)}
|
|
8
|
+
end
|
|
9
|
+
end
|
|
10
|
+
|
|
11
|
+
Gem::Specification.new do |spec|
|
|
12
|
+
spec.name = 'dsv'
|
|
13
|
+
spec.version = DSV::VERSION
|
|
14
|
+
|
|
15
|
+
spec.summary = "Read and write delimiter-separated values."
|
|
16
|
+
spec.description = "Read and write the DSV family of files (CSV, TSV, etc.) from disk or in memory."
|
|
17
|
+
|
|
18
|
+
spec.author = 'thoran'
|
|
19
|
+
spec.email = 'code@thoran.com'
|
|
20
|
+
spec.homepage = "https://github.com/thoran/dsv"
|
|
21
|
+
spec.license = 'MIT'
|
|
22
|
+
|
|
23
|
+
spec.required_ruby_version = '>= 3.2'
|
|
24
|
+
spec.require_paths = ['lib']
|
|
25
|
+
|
|
26
|
+
spec.files = [
|
|
27
|
+
'dsv.gemspec',
|
|
28
|
+
Dir['lib/**/*.rb'],
|
|
29
|
+
Dir['test/**/*.rb'],
|
|
30
|
+
'CHANGELOG',
|
|
31
|
+
'Gemfile',
|
|
32
|
+
'LICENSE',
|
|
33
|
+
'Rakefile',
|
|
34
|
+
'README.md',
|
|
35
|
+
'TODO',
|
|
36
|
+
].flatten
|
|
37
|
+
|
|
38
|
+
spec.development_dependencies = [
|
|
39
|
+
['minitest', '~> 6.0'],
|
|
40
|
+
'minitest-mock',
|
|
41
|
+
'rake',
|
|
42
|
+
]
|
|
43
|
+
end
|
data/lib/DSV/File.rb
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
# DSV/File.rb
|
|
2
|
+
# DSV::File
|
|
3
|
+
|
|
4
|
+
require_relative '../dsv'
|
|
5
|
+
|
|
6
|
+
class DSV
|
|
7
|
+
class File < DSV
|
|
8
|
+
|
|
9
|
+
def initialize(filename, *args)
|
|
10
|
+
@filename = ::File.expand_path(filename)
|
|
11
|
+
@args = args
|
|
12
|
+
super(source, *args)
|
|
13
|
+
end
|
|
14
|
+
|
|
15
|
+
def source
|
|
16
|
+
@source ||= ::File.new(filename, mode, permissions)
|
|
17
|
+
end
|
|
18
|
+
|
|
19
|
+
def mode
|
|
20
|
+
@mode ||= DSV.normalised_mode(options[:mode])
|
|
21
|
+
end
|
|
22
|
+
|
|
23
|
+
# The trailing Hash of the arguments, left in place for the parent to take.
|
|
24
|
+
def options
|
|
25
|
+
@args.last.is_a?(::Hash) ? @args.last : {}
|
|
26
|
+
end
|
|
27
|
+
|
|
28
|
+
def permissions
|
|
29
|
+
@permissions ||= options[:permissions]
|
|
30
|
+
end
|
|
31
|
+
|
|
32
|
+
attr_reader :filename
|
|
33
|
+
|
|
34
|
+
end
|
|
35
|
+
end
|
data/lib/DSV/String.rb
ADDED
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
# DSV/String.rb
|
|
2
|
+
# DSV::String
|
|
3
|
+
|
|
4
|
+
require_relative '../dsv'
|
|
5
|
+
|
|
6
|
+
class DSV
|
|
7
|
+
class String < DSV
|
|
8
|
+
|
|
9
|
+
def initialize(string, *args)
|
|
10
|
+
@string = string
|
|
11
|
+
super(source, *args)
|
|
12
|
+
end
|
|
13
|
+
|
|
14
|
+
def source
|
|
15
|
+
@source ||= StringIO.new(@string)
|
|
16
|
+
end
|
|
17
|
+
|
|
18
|
+
end
|
|
19
|
+
end
|