gman 7.0.6 → 8.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (66) hide show
  1. checksums.yaml +4 -4
  2. data/bin/gman +10 -15
  3. data/bin/gman_filter +1 -1
  4. data/config/domains.txt +21 -408
  5. data/config/vendor/dotgovs.csv +237 -119
  6. data/docs/README.md +3 -1
  7. data/lib/gman/domain_list.rb +1 -1
  8. data/lib/gman/identifier.rb +9 -6
  9. data/lib/gman/importer.rb +1 -1
  10. data/lib/gman/locality.rb +1 -1
  11. data/lib/gman/version.rb +1 -1
  12. data/lib/gman.rb +8 -6
  13. metadata +8 -221
  14. data/.github/CODEOWNERS +0 -3
  15. data/.github/ISSUE_TEMPLATE/bug_report.md +0 -28
  16. data/.github/ISSUE_TEMPLATE/feature_request.md +0 -21
  17. data/.github/config.yml +0 -23
  18. data/.github/dependabot.yml +0 -10
  19. data/.github/funding.yml +0 -1
  20. data/.github/no-response.yml +0 -15
  21. data/.github/release-drafter.yml +0 -4
  22. data/.github/settings.yml +0 -33
  23. data/.github/stale.yml +0 -29
  24. data/.github/workflows/ci.yml +0 -23
  25. data/.github/workflows/clean.yml +0 -31
  26. data/.github/workflows/codeql-analysis.yml +0 -70
  27. data/.github/workflows/validate.yml +0 -30
  28. data/.github/workflows/vendor.yml +0 -29
  29. data/.gitignore +0 -6
  30. data/.rspec +0 -2
  31. data/.rubocop.yml +0 -29
  32. data/.rubocop_todo.yml +0 -84
  33. data/Gemfile +0 -5
  34. data/docs/CODE_OF_CONDUCT.md +0 -46
  35. data/docs/CONTRIBUTING.md +0 -92
  36. data/docs/SECURITY.md +0 -3
  37. data/docs/_config.yml +0 -2
  38. data/gman.gemspec +0 -45
  39. data/script/add +0 -19
  40. data/script/alphabetize +0 -14
  41. data/script/bootstrap +0 -5
  42. data/script/cibuild +0 -8
  43. data/script/console +0 -3
  44. data/script/dedupe +0 -22
  45. data/script/profile +0 -21
  46. data/script/prune +0 -22
  47. data/script/reconcile-us +0 -72
  48. data/script/release +0 -38
  49. data/script/validate-domains +0 -34
  50. data/script/vendor +0 -13
  51. data/script/vendor-federal-de +0 -14
  52. data/script/vendor-gov-list +0 -8
  53. data/script/vendor-public-suffix +0 -28
  54. data/script/vendor-swot +0 -43
  55. data/script/vendor-us +0 -42
  56. data/spec/fixtures/domains.txt +0 -4
  57. data/spec/fixtures/obama.txt +0 -5
  58. data/spec/gman/bin_spec.rb +0 -101
  59. data/spec/gman/country_code_spec.rb +0 -39
  60. data/spec/gman/domain_list_spec.rb +0 -110
  61. data/spec/gman/domains_spec.rb +0 -25
  62. data/spec/gman/identifier_spec.rb +0 -218
  63. data/spec/gman/importer_spec.rb +0 -236
  64. data/spec/gman/locality_spec.rb +0 -24
  65. data/spec/gman_spec.rb +0 -74
  66. data/spec/spec_helper.rb +0 -31
data/script/release DELETED
@@ -1,38 +0,0 @@
1
- #!/bin/sh
2
- # Tag and push a release.
3
-
4
- set -e
5
-
6
- # Make sure we're in the project root.
7
-
8
- cd $(dirname "$0")/..
9
-
10
- # Build a new gem archive.
11
-
12
- rm -rf gman-*.gem
13
- gem build -q gman.gemspec
14
-
15
- # Make sure we're on the master branch.
16
-
17
- (git branch | grep -q '* master') || {
18
- echo "Only release from the master branch."
19
- exit 1
20
- }
21
-
22
- # Figure out what version we're releasing.
23
-
24
- tag=v`ls gman-*.gem | sed 's/^gman-\(.*\)\.gem$/\1/'`
25
-
26
- # Make sure we haven't released this version before.
27
-
28
- git fetch -t origin
29
-
30
- (git tag -l | grep -q "$tag") && {
31
- echo "Whoops, there's already a '${tag}' tag."
32
- exit 1
33
- }
34
-
35
- # Tag it and bag it.
36
-
37
- gem push gman-*.gem && git tag "$tag" &&
38
- git push origin master && git push origin "$tag"
@@ -1,34 +0,0 @@
1
- #!/usr/bin/env ruby
2
-
3
- # ! /usr/bin/env ruby
4
- # frozen_string_literal: true
5
-
6
- #
7
- # Add one or more domains to a given group, running the standard import checks
8
- #
9
- # Usage: script/add [GROUP] [DOMAIN(S)]
10
-
11
- require './lib/gman/importer'
12
- require 'parallel'
13
-
14
- importer = Gman::Importer.new({})
15
- options = { skip_dupe: true, skip_resolve: false }
16
- list_path = File.expand_path '../config/domains.txt', __dir__
17
-
18
- importer.logger.info "Starting list: #{Gman::DomainList.current.count} domains"
19
-
20
- Gman.list.to_h.values.shuffle.each do |domains|
21
- # next if ['non-us gov', 'non-us mil', 'US Federal'].include?(group)
22
-
23
- Parallel.each(domains, progress: "Validating") do |domain|
24
- next if domain.start_with?("!")
25
- next if importer.valid_domain?(domain, options)
26
-
27
- importer.logger.warn "#{domain} is not valid, removing from list"
28
- list = File.read(list_path)
29
- list.gsub!(/^#{Regexp.escape(domain)}$\n/, '')
30
- File.write list_path, list
31
- end
32
- end
33
-
34
- importer.logger.info "Ending list: #{Gman::DomainList.current.count} domains"
data/script/vendor DELETED
@@ -1,13 +0,0 @@
1
- #!/bin/bash
2
- #Runs all vendor scripts except script/vendor-nl
3
-
4
- for file in script/vendor-*; do
5
- if [ "$file" != "script/vendor-nl" ]; then
6
- echo "*************************************"
7
- echo "Vendoring $file"
8
- echo "*************************************"
9
- bundle exec "$file"
10
- fi
11
- done
12
-
13
- bundle exec script/alphabetize
@@ -1,14 +0,0 @@
1
- #! /usr/bin/env ruby
2
- # frozen_string_literal: true
3
-
4
- require 'csv'
5
- require 'open-uri'
6
- require './lib/gman'
7
-
8
- url = 'https://raw.githubusercontent.com/robbi5/german-gov-domains/master/data/domains.csv'
9
-
10
- domains = URI.open(url).read.encode('UTF-8')
11
- domains = CSV.parse(domains, headers: true)
12
- domains = domains.map { |row| row['Domain Name'] }
13
-
14
- Gman::Importer.new('German Federal' => domains).import
@@ -1,8 +0,0 @@
1
- #!/bin/sh
2
- #
3
- # Vendors the full list of US .gov domains from https://github.com/GSA/data
4
- # Usage: script/vendor-gov-list
5
-
6
- # Vendor the last file in the dotgov-domains folder that ends in `-full.csv`
7
- wget https://raw.githubusercontent.com/cisagov/dotgov-data/main/current-full.csv -O ./config/vendor/dotgovs.csv
8
-
@@ -1,28 +0,0 @@
1
- #!/usr/bin/env ruby
2
- # frozen_string_literal: true
3
-
4
- # Propagates an initial list of best-guess government domains
5
-
6
- require 'public_suffix'
7
- require 'yaml'
8
- require_relative '../lib/gman'
9
-
10
- # https://gist.github.com/benbalter/6147066
11
- REGEX = /(\.g[ou]{1,2}(v|b|vt)|\.mil|\.gc|\.fed)(\.[a-z]{2})?$/i.freeze
12
-
13
- domains = []
14
- PublicSuffix::List.default.each do |rule|
15
- domain = nil
16
-
17
- if rule.parts.length == 1
18
- domain = rule.parts.first if REGEX.match?(".#{rule.value}")
19
- elsif REGEX.match?(".#{rule.value}")
20
- domain = rule.parts.pop(2).join('.')
21
- end
22
-
23
- domains.push domain unless domain.nil? || domains.include?(domain)
24
- end
25
-
26
- # NOTE: We want to skip resolution here, because a domain like `gov.sv` may be
27
- # a valid TLD, not have any top-level sites, and we'd still want it listed
28
- Gman::Importer.new('non-us gov' => domains).import(skip_resolve: true)
data/script/vendor-swot DELETED
@@ -1,43 +0,0 @@
1
- #! /usr/bin/env ruby
2
- # frozen_string_literal: true
3
-
4
- #
5
- # Vendors the Swot-maintained list of adademic domains into config/academic.txt
6
- # Source: https://github.com/leereilly/swot/
7
- #
8
- # Usage: script/vendor-swot
9
- #
10
- # Will automatically fetch latest version of the list and merge
11
- # You can check for changes and commit via `git status`
12
- #
13
- # It's also probably a good idea to run `script/ci-build` for good measure
14
- #
15
- # Note: We do this, because as a bajillion individual files, Swot takes up 30MB
16
-
17
- require 'gman'
18
- require 'swot'
19
-
20
- # Generate array of all Swot domains
21
- domains = Swot.all_domains
22
- domains << Swot::ACADEMIC_TLDS
23
-
24
- # Init the importer, builiding a DomainList
25
- group = "Academic domains vendored from Swot v#{Swot::VERSION}"
26
- hash = { group => domains }
27
-
28
- importer = Gman::Importer.new(hash)
29
- importer.logger.info "Importing from Swot v#{Swot::VERSION}"
30
- importer.logger.info "Found #{domains.count} academic domains"
31
-
32
- domain_list = importer.domain_list
33
- domain_list.path = Gman.academic_list_path
34
-
35
- # Cleanup and write
36
- # Note: we're not using the import method, as that assume's we're writing the
37
- # government domain list and would use Swot to ensure domains aren't academic
38
- importer.send :normalize_domains!
39
- domain_list.data[group] << Swot::BLACKLIST.map { |domain| "!#{domain}" }
40
- domain_list.data[group] = domain_list.data[group].flatten
41
- domain_list.write
42
-
43
- importer.logger.info "Vendored #{importer.domain_list.count} academic domains."
data/script/vendor-us DELETED
@@ -1,42 +0,0 @@
1
- #! /usr/bin/env ruby
2
- # frozen_string_literal: true
3
-
4
- #
5
- # Vendors the USA.gov-maintained list of US domains into domains.txt
6
- # Source: https://github.com/GSA-OCSIT/govt-urls
7
- #
8
- # Usage: script/vendor-us
9
- #
10
- # Will automatically fetch latest version of the list and merge
11
- # You can check for changes and commit via `git status`
12
- #
13
- # It's also probably a good idea to run `script/ci-build` for good measure
14
-
15
- require './lib/gman'
16
- require 'open-uri'
17
- require 'csv'
18
-
19
- path = File.expand_path('./vendor-us-tmp.csv')
20
- blacklist = %w[usagovQUASI usagovFEDgov]
21
- source = 'https://raw.githubusercontent.com/GSA/govt-urls/main/1_govt_urls_full.csv'
22
- domains = {}
23
-
24
- begin
25
- raw = URI.open(source).read
26
- File.write(path, raw)
27
- data = CSV.table(path)
28
-
29
- data.each do |domain|
30
- next if domain[:type_of_government] == 'Quasigovernmental'
31
-
32
- group = "US #{domain[:type_of_government]}"
33
- group += " (#{domain[:state]})" if domain[:type_of_government] != 'Federal' && domain[:state]
34
- domains[group] ||= []
35
- domains[group] << domain[:domain_name]
36
- end
37
-
38
- domains.reject! { |g, _| blacklist.include?(g) }
39
- Gman::Importer.new(domains).import
40
- ensure
41
- File.delete(path)
42
- end
@@ -1,4 +0,0 @@
1
- // foo
2
- bar.gov
3
- baz.net
4
- !mail.bar.gov
@@ -1,5 +0,0 @@
1
- barry@dcpchicago.org
2
- prof.obama@uchicago.edu
3
- mr.senator@obama.senate.gov
4
- president@whitehouse.gov
5
- commander.in.chief@us.army.mil
@@ -1,101 +0,0 @@
1
- # frozen_string_literal: true
2
-
3
- RSpec.describe 'Gman bin' do
4
- let(:domain) { 'whitehouse.gov' }
5
- let(:args) { [domain] }
6
- let(:command) { 'gman' }
7
- let(:bin_path) do
8
- File.expand_path "../../bin/#{command}", File.dirname(__FILE__)
9
- end
10
- let(:response_parts) { Open3.capture2e('bundle', 'exec', bin_path, *args) }
11
- let(:output) { response_parts[0] }
12
- let(:status) { response_parts[1] }
13
- let(:exit_code) { status.exitstatus }
14
-
15
- context 'a valid domain' do
16
- it 'parses the domain' do
17
- expect(output).to match('Domain : whitehouse.gov')
18
- end
19
-
20
- it "knows it's valid" do
21
- expect(output).to match('Valid government domain')
22
- expect(exit_code).to be(0)
23
- end
24
-
25
- it 'knows the type' do
26
- expect(output).to match(/federal/i)
27
- end
28
-
29
- it 'knows the agency' do
30
- expect(output).to match('Executive Office of the President')
31
- end
32
-
33
- it 'knows the country' do
34
- expect(output).to match('United States')
35
- end
36
-
37
- it 'knows the city' do
38
- expect(output).to match('Washington')
39
- end
40
-
41
- it 'knows the state' do
42
- expect(output).to match('DC')
43
- end
44
-
45
- it 'colors by default' do
46
- expect(output).to match(/\e\[32m/)
47
- end
48
-
49
- context 'with colorization disabled' do
50
- let(:args) { [domain, '--no-color'] }
51
-
52
- it "doesn't color" do
53
- expect(output).not_to match(/\e\[32m/)
54
- end
55
- end
56
- end
57
-
58
- context 'with no args' do
59
- let(:args) { [] }
60
-
61
- it 'displays the help text' do
62
- expect(output).to match('USAGE')
63
- end
64
- end
65
-
66
- context 'an invalid domain' do
67
- let(:domain) { 'foo.invalid' }
68
-
69
- it 'knows the domain is invalid' do
70
- expect(output).to match('Invalid domain')
71
- expect(exit_code).to be(1)
72
- end
73
- end
74
-
75
- context 'a non-government domain' do
76
- let(:domain) { 'github.com' }
77
-
78
- it "knows it's not a government domain" do
79
- expect(output).to match('Not a government domain')
80
- expect(exit_code).to be(1)
81
- end
82
- end
83
-
84
- context 'filtering' do
85
- let(:command) { 'gman_filter' }
86
- let(:txt_path) do
87
- File.expand_path '../fixtures/obama.txt', File.dirname(__FILE__)
88
- end
89
- let(:args) { [txt_path] }
90
-
91
- it 'returns only government domains' do
92
- expected = <<~EXPECTED
93
- mr.senator@obama.senate.gov
94
- president@whitehouse.gov
95
- commander.in.chief@us.army.mil
96
- EXPECTED
97
-
98
- expect(output).to eql(expected)
99
- end
100
- end
101
- end
@@ -1,39 +0,0 @@
1
- # frozen_string_literal: true
2
-
3
- RSpec.describe 'Gman Country Codes' do
4
- {
5
- 'whitehouse.gov' => 'United States of America',
6
- 'foo.gov.uk' => 'United Kingdom of Great Britain and Northern Ireland',
7
- 'army.mil' => 'United States of America',
8
- 'foo.gc.ca' => 'Canada',
9
- 'foo.eu' => nil
10
- }.each do |domain, expected_country|
11
- context "given #{domain.inspect}" do
12
- subject { Gman.new(domain) }
13
-
14
- let(:country) { subject.country }
15
-
16
- it 'knows the country' do
17
- if expected_country.nil?
18
- expect(country).to be_nil
19
- else
20
- expect(country.name).to eql(expected_country)
21
- end
22
- end
23
-
24
- it 'knows the alpha2' do
25
- expected = case expected_country
26
- when 'United States of America'
27
- 'us'
28
- when 'Canada'
29
- 'ca'
30
- when 'United Kingdom of Great Britain and Northern Ireland'
31
- 'gb'
32
- else
33
- 'eu'
34
- end
35
- expect(subject.alpha2).to eql(expected)
36
- end
37
- end
38
- end
39
- end
@@ -1,110 +0,0 @@
1
- # frozen_string_literal: true
2
-
3
- RSpec.describe Gman::DomainList do
4
- let(:data) { subject.data }
5
- let(:canada) { data['Canada municipal'] }
6
-
7
- %i[path contents data].each do |type|
8
- context "when initialized by #{type}" do
9
- subject do
10
- case type
11
- when :path
12
- described_class.new(path: Gman.list_path)
13
- when :contents
14
- contents = File.read(Gman.list_path)
15
- described_class.new(contents: contents)
16
- when :data
17
- data = described_class.new(path: Gman.list_path).to_h
18
- described_class.new(data: data)
19
- end
20
- end
21
-
22
- it 'stores the init var' do
23
- expect(subject.send(type)).not_to be_nil
24
- end
25
-
26
- it 'returns the domain data' do
27
- expect(data).to have_key('Canada federal')
28
- expect(data.values.flatten).to include('gov')
29
- end
30
-
31
- it 'returns the list contents' do
32
- expect(subject.contents).to match(/^gov$/)
33
- end
34
-
35
- it 'knows the list path' do
36
- expect(subject.path).to eql(Gman.list_path)
37
- end
38
-
39
- it 'returns the PublicSuffix list' do
40
- expect(subject.public_suffix_list).to be_a(PublicSuffix::List)
41
- end
42
-
43
- it 'knows if a domain is valid' do
44
- expect(subject.valid?('whitehouse.gov')).to be(true)
45
- end
46
-
47
- it 'knows if a domain is invalid' do
48
- expect(subject.valid?('example.com')).to be(false)
49
- end
50
-
51
- it 'returns the domain groups' do
52
- expect(subject.groups).to include('Canada federal')
53
- end
54
-
55
- it 'returns the domains' do
56
- expect(subject.domains).to include('gov')
57
- end
58
-
59
- it 'returns the domain count' do
60
- expect(subject.count).to be_a(Integer)
61
- expect(subject.count).to be > 100
62
- end
63
-
64
- it 'alphabetizes the list' do
65
- canada.shuffle!
66
- expect(canada.first).not_to eql('100milehouse.com')
67
- subject.alphabetize
68
- expect(canada.first).to eql('100milehouse.com')
69
- end
70
-
71
- it 'outputs public suffix format' do
72
- expect(subject.to_s).to match("// Canada federal\ncanada.ca\n")
73
- end
74
-
75
- it "finds a domain's parent" do
76
- expect(subject.parent_domain('foo.gov.uk')).to eql('gov.uk')
77
- end
78
-
79
- context 'with the list path stubbed' do
80
- let(:stubbed_file_contents) { File.read(stubbed_list_path) }
81
-
82
- before do
83
- subject.instance_variable_set(:@path, stubbed_list_path)
84
- end
85
-
86
- context 'with list data stubbed' do
87
- before do
88
- subject.data = { 'foo' => ['!mail.bar.gov', 'bar.gov', 'baz.net'] }
89
- end
90
-
91
- context 'alphabetizing' do
92
- before { subject.alphabetize }
93
-
94
- it 'puts exceptions last' do
95
- expect(subject.data['foo'].last).to eql('!mail.bar.gov')
96
- end
97
- end
98
-
99
- context 'writing' do
100
- before { subject.write }
101
-
102
- it 'writes the contents' do
103
- expect(stubbed_file_contents).to match("// foo\nbar.gov\nbaz.net")
104
- end
105
- end
106
- end
107
- end
108
- end
109
- end
110
- end
@@ -1,25 +0,0 @@
1
- # frozen_string_literal: true
2
-
3
- RSpec.describe 'Gman domains' do
4
- let(:resolve_domains?) { ENV['GMAN_RESOLVE_DOMAINS'] == 'true' }
5
- let(:importer) { Gman::Importer.new({}) }
6
- let(:options) { { skip_dupe: true, skip_resolve: !resolve_domains? } }
7
-
8
- Gman.list.to_h.each do |group, domains|
9
- next if ['non-us gov', 'non-us mil', 'US Federal'].include?(group)
10
-
11
- context "the #{group} group" do
12
- it 'only contains valid domains' do
13
- invalid_domains = []
14
-
15
- Parallel.each(domains, in_threads: 4) do |domain|
16
- next if importer.valid_domain?(domain, options)
17
-
18
- invalid_domains.push domain
19
- end
20
-
21
- expect(invalid_domains).to be_empty
22
- end
23
- end
24
- end
25
- end