secscan 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,438 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "json"
4
+ require "pathname"
5
+ require "time"
6
+
7
+ module Secscan
8
+ SCAN_SUFFIXES = %w[.js .jsx .ts .tsx .mjs .cjs .json .env .yaml .yml].freeze
9
+ SKIP_DIR_NAMES = %w[.git node_modules vendor dist build out coverage bower_components .cache].freeze
10
+ MAX_FILE_BYTES = 1_048_576
11
+ API_CALL_RE = /(?:app\.(get|post|put|delete|patch)|router\.(get|post|put|delete|patch)|axios\.(get|post|put|delete|patch)|fetch)\s*\(\s*['"`]([^'"`]+)['"`]/i
12
+ SENSITIVE_PATH_RE = /admin|internal|superadmin|debug|token|actuator|secret/i
13
+
14
+ Finding = Struct.new(
15
+ :id, :rule_id, :rule_name, :category, :severity, :file, :line, :column,
16
+ :snippet, :matched_secret, :masked_secret, :entropy, :description,
17
+ :remediation, :file_criticality, :file_criticality_weight, :weighted_score,
18
+ keyword_init: true
19
+ ) do
20
+ def public_snippet(reveal = false)
21
+ return snippet if reveal || matched_secret.to_s.empty?
22
+
23
+ snippet.gsub(matched_secret, masked_secret)
24
+ end
25
+ end
26
+
27
+ ApiEndpoint = Struct.new(
28
+ :id, :file, :line, :method, :path, :is_internal_or_admin, :snippet,
29
+ keyword_init: true
30
+ )
31
+
32
+ Metrics = Struct.new(
33
+ :critical_count, :high_count, :medium_count, :low_count, :info_count,
34
+ :security_score, :security_impact_score, :impact_level, :total_weighted_risk,
35
+ :average_entropy,
36
+ keyword_init: true
37
+ )
38
+
39
+ ScanReport = Struct.new(
40
+ :scanner, :version, :timestamp, :target, :total_files, :scanned_files_count,
41
+ :ignored_files_count, :findings, :api_endpoints, :metrics, :duration_ms,
42
+ keyword_init: true
43
+ )
44
+
45
+ module Scanner
46
+ module_function
47
+
48
+ def calculate_entropy(value)
49
+ return 0.0 if value.nil? || value.empty?
50
+
51
+ length = value.length.to_f
52
+ entropy = value.each_char.tally.each_value.sum do |count|
53
+ probability = count / length
54
+ -probability * Math.log2(probability)
55
+ end
56
+ format("%.2f", entropy).to_f
57
+ end
58
+
59
+ def mask_secret(secret)
60
+ return "" if secret.nil? || secret.empty?
61
+
62
+ clean = secret.strip.gsub(/\A['"]|['"]\z/, "")
63
+ return "••••••••" if clean.length <= 8
64
+
65
+ hidden = [16, [6, clean.length - 8].max].min
66
+ "#{clean[0, 4]}#{'•' * hidden}#{clean[-4, 4]}"
67
+ end
68
+
69
+ def scan_path(path, rules: nil, ignore: [])
70
+ root = File.expand_path(path.to_s)
71
+ raise InputError, "path not found: #{path}" unless File.exist?(root)
72
+
73
+ files, seen, ignored = collect_files(root, ignore)
74
+ scan_files(files, rules: rules, ignore: ignore, target: path.to_s, ignored_files_count: ignored, total_files: seen)
75
+ end
76
+
77
+ def scan_text(content, path: "snippet", rules: nil, ignore: [])
78
+ scan_files([[path, content]], rules: rules, ignore: ignore, target: path)
79
+ end
80
+
81
+ def load_rules(path = nil, replace: false)
82
+ rules = replace ? [] : DEFAULT_RULES.dup
83
+ return rules.select(&:enabled) if path.nil?
84
+
85
+ raise InputError, "rules file not found: #{path}" unless File.file?(path)
86
+
87
+ begin
88
+ payload = JSON.parse(File.read(path, encoding: "UTF-8"))
89
+ rescue JSON::ParserError
90
+ raise InputError, "rules file is not valid JSON: #{path}"
91
+ end
92
+ raise InputError, "rules file must be a JSON array" unless payload.is_a?(Array)
93
+
94
+ payload.each do |item|
95
+ raise InputError, "each rule must be a JSON object" unless item.is_a?(Hash)
96
+
97
+ begin
98
+ min_entropy = item.key?("minEntropy") ? item["minEntropy"] : item["min_entropy"]
99
+ rules << Rule.new(
100
+ id: item.fetch("id"),
101
+ name: item.fetch("name"),
102
+ pattern: item.fetch("pattern"),
103
+ severity: item["severity"] || "MEDIUM",
104
+ category: item["category"] || "CUSTOM",
105
+ description: item["description"] || "",
106
+ remediation: item["remediation"] || "",
107
+ flags: item["flags"] || "g",
108
+ min_entropy: min_entropy.nil? ? nil : min_entropy.to_f,
109
+ enabled: item.fetch("enabled", true)
110
+ )
111
+ rescue ArgumentError, KeyError => e
112
+ raise InputError, "invalid rule: #{e.message}"
113
+ end
114
+ end
115
+ rules.select(&:enabled)
116
+ end
117
+
118
+ def scan_files(files, rules: nil, ignore: [], target: "memory", ignored_files_count: 0, total_files: nil)
119
+ started = Process.clock_gettime(Process::CLOCK_MONOTONIC)
120
+ active = rules.nil? ? DEFAULT_RULES : rules
121
+ compiled = active.map { |rule| [rule, compile_rule(rule)] }
122
+ findings = []
123
+ endpoints = []
124
+ scanned = 0
125
+ extra_ignored = 0
126
+
127
+ files.each do |rel_path, content|
128
+ if ignore_reason(rel_path, ignore)
129
+ extra_ignored += 1
130
+ next
131
+ end
132
+ scanned += 1
133
+ level, weight, = evaluate_file_criticality(rel_path)
134
+
135
+ compiled.each do |rule, pattern|
136
+ next if pattern.nil?
137
+
138
+ content.scan(pattern) do
139
+ match = Regexp.last_match
140
+ literal = match[0]
141
+ next if literal.nil? || literal.empty?
142
+
143
+ entropy = calculate_entropy(literal)
144
+ next if !rule.min_entropy.nil? && entropy < rule.min_entropy
145
+
146
+ line, column, snippet = line_column(content, match.begin(0))
147
+ base = SEVERITY_BASE_WEIGHTS.fetch(rule.severity, 5)
148
+ findings << Finding.new(
149
+ id: "finding-#{findings.length + 1}",
150
+ rule_id: rule.id,
151
+ rule_name: rule.name,
152
+ category: rule.category,
153
+ severity: rule.severity,
154
+ file: rel_path,
155
+ line: line,
156
+ column: column,
157
+ snippet: snippet,
158
+ matched_secret: literal,
159
+ masked_secret: mask_secret(literal),
160
+ entropy: entropy,
161
+ description: rule.description,
162
+ remediation: rule.remediation,
163
+ file_criticality: level,
164
+ file_criticality_weight: weight,
165
+ weighted_score: format("%.1f", base * weight).to_f
166
+ )
167
+ end
168
+ end
169
+
170
+ content.scan(API_CALL_RE) do
171
+ match = Regexp.last_match
172
+ raw_path = match[4].to_s
173
+ next unless raw_path.start_with?("/") || raw_path.start_with?("http")
174
+
175
+ method = (match[1] || match[2] || match[3] || "GET").upcase
176
+ line, = line_column(content, match.begin(0))
177
+ endpoints << ApiEndpoint.new(
178
+ id: "endpoint-#{endpoints.length + 1}",
179
+ file: rel_path,
180
+ line: line,
181
+ method: method,
182
+ path: raw_path,
183
+ is_internal_or_admin: SENSITIVE_PATH_RE.match?(raw_path),
184
+ snippet: line_column(content, match.begin(0))[2]
185
+ )
186
+ end
187
+ end
188
+
189
+ counts = SEVERITY_RANK.keys.to_h { |name| [name, 0] }
190
+ findings.each { |finding| counts[finding.severity] += 1 }
191
+ deduction = (counts["CRITICAL"] * 25) + (counts["HIGH"] * 12) + (counts["MEDIUM"] * 5) + (counts["LOW"] * 2)
192
+ security_score = [[100 - deduction, 0].max, 100].min
193
+ impact, weighted, impact_level = calculate_security_impact(findings)
194
+ average = findings.empty? ? 0.0 : format("%.2f", findings.sum(&:entropy) / findings.length.to_f).to_f
195
+ elapsed = ((Process.clock_gettime(Process::CLOCK_MONOTONIC) - started) * 1000).round
196
+
197
+ ScanReport.new(
198
+ scanner: "SecScan SAST",
199
+ version: VERSION,
200
+ timestamp: Time.now.utc.iso8601,
201
+ target: target,
202
+ total_files: total_files.nil? ? files.length : total_files,
203
+ scanned_files_count: scanned,
204
+ ignored_files_count: ignored_files_count + extra_ignored,
205
+ findings: findings,
206
+ api_endpoints: endpoints,
207
+ metrics: Metrics.new(
208
+ critical_count: counts["CRITICAL"],
209
+ high_count: counts["HIGH"],
210
+ medium_count: counts["MEDIUM"],
211
+ low_count: counts["LOW"],
212
+ info_count: counts["INFO"],
213
+ security_score: security_score,
214
+ security_impact_score: impact,
215
+ impact_level: impact_level,
216
+ total_weighted_risk: weighted,
217
+ average_entropy: average
218
+ ),
219
+ duration_ms: elapsed
220
+ )
221
+ end
222
+
223
+ def calculate_security_impact(findings)
224
+ return [0, 0.0, "NOMINAL"] if findings.empty?
225
+
226
+ total = format("%.1f", findings.sum(&:weighted_score)).to_f
227
+ normalized = 100 * (1 - Math.exp(-total / 55.0))
228
+ score = [[normalized.round, 1].max, 100].min
229
+ level = if score >= 80
230
+ "CRITICAL"
231
+ elsif score >= 60
232
+ "HIGH"
233
+ elsif score >= 35
234
+ "ELEVATED"
235
+ elsif score >= 15
236
+ "MODERATE"
237
+ elsif score.positive?
238
+ "LOW"
239
+ else
240
+ "NOMINAL"
241
+ end
242
+ [score, total, level]
243
+ end
244
+
245
+ def evaluate_file_criticality(file_path)
246
+ normalized = file_path.to_s.tr("\\", "/").downcase
247
+ file_name = normalized.split("/").last.to_s
248
+
249
+ if normalized.include?(".env") || normalized.include?("credentials") || normalized.include?("secret") ||
250
+ normalized.include?("id_rsa") || file_name.end_with?(".pem", ".key", ".pfx", ".keystore") ||
251
+ normalized.start_with?("config/") || normalized.include?("/config/") ||
252
+ file_name.include?("cloudconfig") || file_name.include?("dbconfig") || file_name.include?("database") ||
253
+ file_name == "dockerfile" || file_name.start_with?("docker-compose") ||
254
+ normalized.include?("k8s/") || normalized.include?("kubernetes/") || normalized.include?("helm/") ||
255
+ %w[server.ts server.js].include?(file_name)
256
+ return ["CRITICAL", FILE_CRITICALITY_MULTIPLIERS["CRITICAL"], "CONFIG_INFRA_SECRETS"]
257
+ end
258
+
259
+ if normalized.include?("/services/") || normalized.include?("/controllers/") || normalized.include?("/routes/") ||
260
+ normalized.include?("/api/") || normalized.include?("/handlers/") || normalized.include?("/backend/") ||
261
+ normalized.include?("/auth") || normalized.include?("payment") || normalized.include?("checkout") ||
262
+ normalized.include?("webhook") || file_name.include?("authservice") || file_name.include?("paymentcontroller") ||
263
+ normalized.include?("firebase.json") || normalized.include?("cloudbuild") || normalized.include?("terraform")
264
+ return ["HIGH", FILE_CRITICALITY_MULTIPLIERS["HIGH"], "BACKEND_API_SERVICE"]
265
+ end
266
+
267
+ if normalized.include?("/test/") || normalized.include?("/tests/") || normalized.include?("/__tests__/") ||
268
+ normalized.include?("/mocks/") || normalized.include?("/fixtures/") || normalized.include?("/docs/") ||
269
+ file_name.end_with?(".test.ts", ".test.js", ".spec.ts", ".spec.js", ".md", ".txt", ".css", ".svg")
270
+ return ["LOW", FILE_CRITICALITY_MULTIPLIERS["LOW"], "TEST_DOC_FIXTURE"]
271
+ end
272
+
273
+ ["MEDIUM", FILE_CRITICALITY_MULTIPLIERS["MEDIUM"], "APPLICATION_CLIENT"]
274
+ end
275
+
276
+ def matches_ignore_pattern(file_path, raw_pattern)
277
+ return false if raw_pattern.nil? || raw_pattern.empty? || file_path.nil? || file_path.empty?
278
+
279
+ pattern = raw_pattern.strip.tr("\\", "/")
280
+ return false if pattern.empty? || pattern.start_with?("#")
281
+
282
+ normalized_path = normalize_rel(file_path)
283
+ normalized_pattern = normalize_rel(pattern)
284
+ return true if normalized_path == normalized_pattern
285
+
286
+ if normalized_pattern.end_with?("/*", "/**", "/")
287
+ folder = normalized_pattern.sub(%r{(/\*+|/)\z}, "")
288
+ return true if normalized_path == folder || normalized_path.start_with?("#{folder}/") ||
289
+ "/#{normalized_path}".include?("/#{folder}/")
290
+ end
291
+
292
+ if !normalized_pattern.include?("*") && !normalized_pattern.include?(".")
293
+ return true if normalized_path == normalized_pattern || normalized_path.start_with?("#{normalized_pattern}/") ||
294
+ "/#{normalized_path}".include?("/#{normalized_pattern}/") ||
295
+ normalized_path.end_with?("/#{normalized_pattern}")
296
+ end
297
+
298
+ if normalized_pattern.start_with?("*.")
299
+ suffix = normalized_pattern[1..]
300
+ if suffix.end_with?(".*")
301
+ base = suffix[0..-3]
302
+ return true if normalized_path.include?("#{base}.") || normalized_path.end_with?(base)
303
+ elsif normalized_path.end_with?(suffix)
304
+ return true
305
+ end
306
+ end
307
+
308
+ glob_match?(normalized_path, normalized_pattern)
309
+ end
310
+
311
+ def ignore_reason(file_path, custom_patterns = [])
312
+ Array(custom_patterns).each do |pattern|
313
+ return "ignore:#{pattern}" if matches_ignore_pattern(file_path, pattern)
314
+ end
315
+
316
+ normalized = file_path.to_s.tr("\\", "/").downcase
317
+ padded = "/#{normalized}"
318
+ return "node_modules" if padded.include?("/node_modules/") || normalized.start_with?("node_modules/")
319
+ return "vendor" if padded.include?("/vendor/") || normalized.start_with?("vendor/")
320
+ return "bower_components" if padded.include?("/bower_components/")
321
+ return "git" if padded.include?("/.git/") || normalized.start_with?(".git/")
322
+ return "build" if %w[/dist/ /build/ /out/].any? { |token| padded.include?(token) } ||
323
+ normalized.start_with?("dist/", "build/", "out/")
324
+ return "bundle" if normalized.end_with?(".min.js", ".bundle.js") || normalized.include?(".chunk.js")
325
+ return "lockfile" if normalized.end_with?("package-lock.json", "yarn.lock", "pnpm-lock.yaml")
326
+ return "cache" if padded.include?("/coverage/") || padded.include?("/.cache/")
327
+
328
+ nil
329
+ end
330
+
331
+ def normalize_rel(value)
332
+ text = value.to_s.strip.tr("\\", "/")
333
+ text = text[2..] if text.start_with?("./")
334
+ text = text[1..] if text.start_with?("/")
335
+ text.downcase
336
+ end
337
+
338
+ def glob_match?(path, pattern)
339
+ parts = []
340
+ index = 0
341
+ while index < pattern.length
342
+ if pattern[index, 2] == "**"
343
+ parts << ".*"
344
+ index += 2
345
+ elsif pattern[index] == "*"
346
+ parts << "[^/]*"
347
+ index += 1
348
+ elsif pattern[index] == "?"
349
+ parts << "."
350
+ index += 1
351
+ else
352
+ parts << Regexp.escape(pattern[index])
353
+ index += 1
354
+ end
355
+ end
356
+ /(?:^|\/)#{parts.join}(?:$|\/)/.match?(path)
357
+ end
358
+
359
+ def compile_rule(rule)
360
+ pattern = rule.pattern.to_s.dup
361
+ flags = rule.flags.to_s
362
+ if pattern.start_with?("(?i)")
363
+ pattern = pattern[4..]
364
+ flags += "i" unless flags.include?("i")
365
+ end
366
+ options = flags.include?("i") ? Regexp::IGNORECASE : 0
367
+ Regexp.new(pattern, options)
368
+ rescue RegexpError
369
+ nil
370
+ end
371
+
372
+ def line_column(content, index)
373
+ prefix = content[0...index].to_s
374
+ line = prefix.count("\n") + 1
375
+ column = prefix.split("\n", -1).last.to_s.length + 1
376
+ snippet = content.split("\n", -1)[line - 1].to_s.strip
377
+ [line, column, snippet]
378
+ end
379
+
380
+ def scannable?(path)
381
+ name = File.basename(path)
382
+ return true if name == ".env" || name.start_with?(".env.")
383
+
384
+ SCAN_SUFFIXES.include?(File.extname(name).downcase)
385
+ end
386
+
387
+ def collect_files(root, custom_ignores)
388
+ return collect_one(root, custom_ignores) if File.file?(root)
389
+
390
+ files = []
391
+ seen = 0
392
+ ignored = 0
393
+ Dir.glob(File.join(root, "**", "*"), File::FNM_DOTMATCH).each do |full|
394
+ next unless File.file?(full)
395
+
396
+ relative = Pathname.new(full).relative_path_from(Pathname.new(root)).to_s.tr("\\", "/")
397
+ parts = relative.split("/")
398
+ next if parts.any? { |part| SKIP_DIR_NAMES.include?(part) }
399
+ next unless scannable?(full)
400
+
401
+ seen += 1
402
+ if ignore_reason(relative, custom_ignores)
403
+ ignored += 1
404
+ next
405
+ end
406
+ text = read_text(full)
407
+ if text.nil?
408
+ ignored += 1
409
+ next
410
+ end
411
+ files << [relative, text]
412
+ end
413
+ [files, seen, ignored]
414
+ end
415
+
416
+ def collect_one(root, custom_ignores)
417
+ relative = root.tr("\\", "/")
418
+ return [[], 1, 1] if ignore_reason(relative, custom_ignores) || !scannable?(root)
419
+
420
+ text = read_text(root)
421
+ return [[], 1, 1] if text.nil?
422
+
423
+ [[relative, text], 1, 0]
424
+ end
425
+
426
+ def read_text(path)
427
+ return nil if File.size(path) > MAX_FILE_BYTES
428
+
429
+ data = File.binread(path)
430
+ return nil if data.include?("\0")
431
+
432
+ data.force_encoding(Encoding::UTF_8)
433
+ data.valid_encoding? ? data : data.encode("UTF-8", invalid: :replace, undef: :replace)
434
+ rescue SystemCallError
435
+ nil
436
+ end
437
+ end
438
+ end
@@ -0,0 +1,5 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Secscan
4
+ VERSION = "0.1.0"
5
+ end
data/lib/secscan.rb ADDED
@@ -0,0 +1,30 @@
1
+ # frozen_string_literal: true
2
+
3
+ require_relative "secscan/version"
4
+ require_relative "secscan/errors"
5
+ require_relative "secscan/rules"
6
+ require_relative "secscan/scanner"
7
+ require_relative "secscan/report"
8
+ require_relative "secscan/cli"
9
+
10
+ module Secscan
11
+ def self.scan(path, **options)
12
+ Scanner.scan_path(path, **options)
13
+ end
14
+
15
+ def self.scan_text(content, **options)
16
+ Scanner.scan_text(content, **options)
17
+ end
18
+
19
+ def self.load_rules(path = nil, **options)
20
+ Scanner.load_rules(path, **options)
21
+ end
22
+
23
+ def self.calculate_entropy(value)
24
+ Scanner.calculate_entropy(value)
25
+ end
26
+
27
+ def self.mask_secret(secret)
28
+ Scanner.mask_secret(secret)
29
+ end
30
+ end
metadata ADDED
@@ -0,0 +1,90 @@
1
+ --- !ruby/object:Gem::Specification
2
+ name: secscan
3
+ version: !ruby/object:Gem::Version
4
+ version: 0.1.0
5
+ platform: ruby
6
+ authors:
7
+ - Saulo Filho
8
+ bindir: exe
9
+ cert_chain: []
10
+ date: 1980-01-02 00:00:00.000000000 Z
11
+ dependencies:
12
+ - !ruby/object:Gem::Dependency
13
+ name: rake
14
+ requirement: !ruby/object:Gem::Requirement
15
+ requirements:
16
+ - - "~>"
17
+ - !ruby/object:Gem::Version
18
+ version: '13.0'
19
+ type: :development
20
+ prerelease: false
21
+ version_requirements: !ruby/object:Gem::Requirement
22
+ requirements:
23
+ - - "~>"
24
+ - !ruby/object:Gem::Version
25
+ version: '13.0'
26
+ - !ruby/object:Gem::Dependency
27
+ name: rspec
28
+ requirement: !ruby/object:Gem::Requirement
29
+ requirements:
30
+ - - "~>"
31
+ - !ruby/object:Gem::Version
32
+ version: '3.12'
33
+ type: :development
34
+ prerelease: false
35
+ version_requirements: !ruby/object:Gem::Requirement
36
+ requirements:
37
+ - - "~>"
38
+ - !ruby/object:Gem::Version
39
+ version: '3.12'
40
+ description: |
41
+ Static analysis engine for JavaScript and TypeScript trees. Detects hardcoded
42
+ secrets, high-entropy tokens, and sensitive API paths, then emits table, JSON,
43
+ CSV, SARIF, or Markdown. Fails CI when a severity threshold or cumulative
44
+ impact score is exceeded. Serialized reports mask matched values.
45
+ email:
46
+ - saulofilho@users.noreply.github.com
47
+ executables:
48
+ - secscan
49
+ extensions: []
50
+ extra_rdoc_files: []
51
+ files:
52
+ - CHANGELOG.md
53
+ - LICENSE.txt
54
+ - README.md
55
+ - examples/poc/custom-rules.json
56
+ - examples/poc/src/app.js
57
+ - exe/secscan
58
+ - lib/secscan.rb
59
+ - lib/secscan/cli.rb
60
+ - lib/secscan/errors.rb
61
+ - lib/secscan/report.rb
62
+ - lib/secscan/rules.rb
63
+ - lib/secscan/scanner.rb
64
+ - lib/secscan/version.rb
65
+ homepage: https://saulofilho.github.io/secscan/
66
+ licenses:
67
+ - MIT
68
+ metadata:
69
+ homepage_uri: https://saulofilho.github.io/secscan/
70
+ source_code_uri: https://github.com/saulofilho/secscan-ruby
71
+ changelog_uri: https://github.com/saulofilho/secscan-ruby/blob/main/CHANGELOG.md
72
+ rubygems_mfa_required: 'true'
73
+ rdoc_options: []
74
+ require_paths:
75
+ - lib
76
+ required_ruby_version: !ruby/object:Gem::Requirement
77
+ requirements:
78
+ - - ">="
79
+ - !ruby/object:Gem::Version
80
+ version: 3.1.0
81
+ required_rubygems_version: !ruby/object:Gem::Requirement
82
+ requirements:
83
+ - - ">="
84
+ - !ruby/object:Gem::Version
85
+ version: '0'
86
+ requirements: []
87
+ rubygems_version: 4.0.16
88
+ specification_version: 4
89
+ summary: SAST engine for secrets, Shannon entropy, API paths, and CI quality gates
90
+ test_files: []