simple_english 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +7 -0
- data/LICENSE +21 -0
- data/README.md +193 -0
- data/bin/se +6 -0
- data/docs/RULES.md +496 -0
- data/lib/simple_english/annotated_text.rb +117 -0
- data/lib/simple_english/cli.rb +152 -0
- data/lib/simple_english/client.rb +133 -0
- data/lib/simple_english/config.rb +51 -0
- data/lib/simple_english/counts.rb +40 -0
- data/lib/simple_english/engine.rb +31 -0
- data/lib/simple_english/extractor.rb +55 -0
- data/lib/simple_english/finding.rb +8 -0
- data/lib/simple_english/http.rb +76 -0
- data/lib/simple_english/install.rb +75 -0
- data/lib/simple_english/languagetool.rb +143 -0
- data/lib/simple_english/markdown.rb +97 -0
- data/lib/simple_english/paragraph.rb +10 -0
- data/lib/simple_english/plain_text.rb +17 -0
- data/lib/simple_english/result.rb +18 -0
- data/lib/simple_english/segment.rb +11 -0
- data/lib/simple_english/server.rb +196 -0
- data/lib/simple_english/span.rb +8 -0
- data/lib/simple_english/suppressions.rb +37 -0
- data/lib/simple_english.rb +78 -0
- data/rules/simple-english.xml +617 -0
- metadata +168 -0
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
# LanguageTool AnnotatedText for code comments. Builds the annotation
|
|
4
|
+
# JSON and maps LT match offsets back to file line and UTF-16 column.
|
|
5
|
+
# Offsets are UTF-16 code units that count into the concatenation of
|
|
6
|
+
# all text and markup strings. interpretAs does not count. Semantics
|
|
7
|
+
# verified against a live LT 6.6 server on 2026-09-25.
|
|
8
|
+
|
|
9
|
+
require_relative "extractor"
|
|
10
|
+
require_relative "segment"
|
|
11
|
+
require_relative "result"
|
|
12
|
+
|
|
13
|
+
module SimpleEnglish
|
|
14
|
+
module AnnotatedText
|
|
15
|
+
# ASCII comment markers only (MARKER holds the set).
|
|
16
|
+
MARKER = /\A[\/#*;%-]+/
|
|
17
|
+
CLOSER = /\A.*?(\s*\*\/)\z/m
|
|
18
|
+
|
|
19
|
+
module_function
|
|
20
|
+
|
|
21
|
+
def build(source, spans)
|
|
22
|
+
segments = []
|
|
23
|
+
cursor = 0 # byte offset into source
|
|
24
|
+
emit = lambda do |text:, content:, file_char:, interpret_as: nil|
|
|
25
|
+
next if content.empty?
|
|
26
|
+
segments << Segment.new(text: text, content: content, file_char: file_char,
|
|
27
|
+
stream_start: segments.sum { |s| s.content.length }, interpret_as: interpret_as)
|
|
28
|
+
end
|
|
29
|
+
spans = spans.sort_by(&:start_byte)
|
|
30
|
+
spans.each do |span|
|
|
31
|
+
if cursor < span.start_byte
|
|
32
|
+
# Interior gaps break sentences between comments. Gaps with no
|
|
33
|
+
# checkable text before them (leading) change nothing.
|
|
34
|
+
emit.call(text: false, content: source.byteslice(cursor...span.start_byte),
|
|
35
|
+
file_char: char_index(source, cursor),
|
|
36
|
+
interpret_as: (segments.any? { |s| s.text }) ? "\n\n" : nil)
|
|
37
|
+
end
|
|
38
|
+
marker, body, closer = split(span.text)
|
|
39
|
+
file_char = char_index(source, span.start_byte)
|
|
40
|
+
emit.call(text: false, content: marker, file_char: file_char, interpret_as: " ")
|
|
41
|
+
emit.call(text: true, content: body, file_char: file_char + marker.length)
|
|
42
|
+
unless closer.empty?
|
|
43
|
+
emit.call(text: false, content: closer,
|
|
44
|
+
file_char: file_char + marker.length + body.length)
|
|
45
|
+
end
|
|
46
|
+
cursor = span.end_byte
|
|
47
|
+
end
|
|
48
|
+
emit.call(text: false, content: source.byteslice(cursor..),
|
|
49
|
+
file_char: char_index(source, cursor), interpret_as: nil)
|
|
50
|
+
Result.new(source: source, segments: segments,
|
|
51
|
+
stream: segments.map(&:content).join)
|
|
52
|
+
end
|
|
53
|
+
|
|
54
|
+
def data_json(result)
|
|
55
|
+
require "json"
|
|
56
|
+
JSON.generate("annotation" => result.segments.map do |segment|
|
|
57
|
+
if segment.text
|
|
58
|
+
{"text" => segment.content}
|
|
59
|
+
else
|
|
60
|
+
entry = {"markup" => segment.content}
|
|
61
|
+
entry["interpretAs"] = segment.interpret_as if segment.interpret_as
|
|
62
|
+
entry
|
|
63
|
+
end
|
|
64
|
+
end)
|
|
65
|
+
end
|
|
66
|
+
|
|
67
|
+
# -> [line, column], both 1-based. Column in UTF-16 code units.
|
|
68
|
+
def locate(result, utf16_offset)
|
|
69
|
+
char = utf16_to_char(result.stream, utf16_offset)
|
|
70
|
+
segment = result.segments.find do |s|
|
|
71
|
+
char >= s.stream_start && char < s.stream_start + s.content.length
|
|
72
|
+
end || result.segments.last
|
|
73
|
+
file_char = segment.file_char + (char - segment.stream_start)
|
|
74
|
+
starts = line_starts(result.source)
|
|
75
|
+
line = starts.rindex { |start| start <= file_char } + 1
|
|
76
|
+
[line, utf16_column(result.source, starts[line - 1], file_char)]
|
|
77
|
+
end
|
|
78
|
+
|
|
79
|
+
def split(comment)
|
|
80
|
+
marker = comment[MARKER] || ""
|
|
81
|
+
body = comment[marker.length..] || comment
|
|
82
|
+
closer = body[CLOSER] || ""
|
|
83
|
+
body = body[0, body.length - closer.length] unless closer.empty?
|
|
84
|
+
[marker, body, closer]
|
|
85
|
+
end
|
|
86
|
+
|
|
87
|
+
def char_index(source, byte)
|
|
88
|
+
# ponytail: O(filesize) per call. If large repos make this
|
|
89
|
+
# measurable, index byte->char once.
|
|
90
|
+
source.byteslice(0...byte).length
|
|
91
|
+
end
|
|
92
|
+
|
|
93
|
+
def line_starts(source)
|
|
94
|
+
starts = [0]
|
|
95
|
+
source.chars.each_with_index { |c, i| starts << i + 1 if c == "\n" }
|
|
96
|
+
starts
|
|
97
|
+
end
|
|
98
|
+
|
|
99
|
+
def utf16_to_char(stream, offset)
|
|
100
|
+
units = 0
|
|
101
|
+
stream.each_char.with_index do |char, index|
|
|
102
|
+
return index if units >= offset
|
|
103
|
+
units += (char.ord > 0xFFFF) ? 2 : 1
|
|
104
|
+
end
|
|
105
|
+
stream.length
|
|
106
|
+
end
|
|
107
|
+
|
|
108
|
+
def utf16_column(source, from, to)
|
|
109
|
+
units = 1
|
|
110
|
+
source[from...to].each_char { |c| units += (c.ord > 0xFFFF) ? 2 : 1 }
|
|
111
|
+
units
|
|
112
|
+
end
|
|
113
|
+
|
|
114
|
+
private_class_method :split, :char_index, :line_starts,
|
|
115
|
+
:utf16_to_char, :utf16_column
|
|
116
|
+
end
|
|
117
|
+
end
|
|
@@ -0,0 +1,152 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
# The command-line interface: Thor commands for lint, serve, and
|
|
4
|
+
# setup. bin/se is a thin runner over this, so the installed
|
|
5
|
+
# gem's RubyGems shim can load it.
|
|
6
|
+
|
|
7
|
+
require "thor"
|
|
8
|
+
require_relative "../simple_english"
|
|
9
|
+
|
|
10
|
+
module SimpleEnglish
|
|
11
|
+
class CLI < Thor
|
|
12
|
+
def self.exit_on_failure? = true
|
|
13
|
+
|
|
14
|
+
# Bare file arguments and "-" (stdin) are lint targets, not command
|
|
15
|
+
# names. Thor only falls back to the default task when the first
|
|
16
|
+
# argument is an option, so route them to lint explicitly.
|
|
17
|
+
def self.start(given_args = ARGV, config = {})
|
|
18
|
+
first = given_args.first
|
|
19
|
+
if first && !all_commands.key?(first) &&
|
|
20
|
+
(first == "-" || !first.start_with?("-"))
|
|
21
|
+
given_args = ["lint", *given_args]
|
|
22
|
+
end
|
|
23
|
+
super
|
|
24
|
+
end
|
|
25
|
+
|
|
26
|
+
# The test entry point. It keeps the old module interface.
|
|
27
|
+
def self.run(argv)
|
|
28
|
+
start(argv)
|
|
29
|
+
end
|
|
30
|
+
|
|
31
|
+
default_task :lint
|
|
32
|
+
desc "lint FILE_OR_DIR...", "Lint Markdown prose and code comments (- reads stdin)"
|
|
33
|
+
method_option :format, type: :string, default: "text", enum: %w[text json sarif],
|
|
34
|
+
banner: "text|json|sarif"
|
|
35
|
+
def lint(*paths)
|
|
36
|
+
if paths.empty?
|
|
37
|
+
help("lint")
|
|
38
|
+
return 2
|
|
39
|
+
end
|
|
40
|
+
config =
|
|
41
|
+
begin
|
|
42
|
+
SimpleEnglish::Config.load
|
|
43
|
+
rescue SimpleEnglish::Config::ConfigError => e
|
|
44
|
+
warn "error: #{e.message}"
|
|
45
|
+
return 2
|
|
46
|
+
end
|
|
47
|
+
results = []
|
|
48
|
+
self.class.expand_paths(paths).each do |path|
|
|
49
|
+
next if path != "-" && SimpleEnglish::Config.ignore?(config, path)
|
|
50
|
+
unless path == "-" || File.readable?(path)
|
|
51
|
+
warn "error: file not readable: #{path}"
|
|
52
|
+
return 2
|
|
53
|
+
end
|
|
54
|
+
text = (path == "-") ? $stdin.read : File.read(path)
|
|
55
|
+
findings = SimpleEnglish.lint_file(path, text)
|
|
56
|
+
return 2 if findings.nil? # already warned why
|
|
57
|
+
SimpleEnglish::Config.filter(config, path, findings).each do |finding|
|
|
58
|
+
results << [path, finding]
|
|
59
|
+
end
|
|
60
|
+
end
|
|
61
|
+
self.class.report(results, options[:format])
|
|
62
|
+
results.empty? ? 0 : 1
|
|
63
|
+
end
|
|
64
|
+
|
|
65
|
+
desc "serve", "Run the lint daemon in the foreground"
|
|
66
|
+
method_option :port, type: :numeric, default: SimpleEnglish::Client::DEFAULT_PORT,
|
|
67
|
+
desc: "Port to listen on"
|
|
68
|
+
method_option :"dev-log", type: :string, banner: "PATH",
|
|
69
|
+
desc: "Write daemon stderr to PATH"
|
|
70
|
+
def serve
|
|
71
|
+
log = options[:"dev-log"] ? File.open(options[:"dev-log"], "w") : $stderr
|
|
72
|
+
SimpleEnglish::Server.start(port: options[:port],
|
|
73
|
+
install: SimpleEnglish::Install.from_env, log: log)
|
|
74
|
+
0
|
|
75
|
+
rescue SimpleEnglish::Install::SetupError, SimpleEnglish::Server::ServerError => e
|
|
76
|
+
warn "error: #{e.message}"
|
|
77
|
+
2
|
|
78
|
+
end
|
|
79
|
+
|
|
80
|
+
desc "setup", "Download LanguageTool and locate Java. Idempotent."
|
|
81
|
+
method_option :dir, type: :string, banner: "PATH",
|
|
82
|
+
desc: "Install into PATH (default: the shared cache)"
|
|
83
|
+
def setup
|
|
84
|
+
install = SimpleEnglish::Install.from_env
|
|
85
|
+
lt_dir = SimpleEnglish::LanguageTool.install(options[:dir] || install.cache_dir)
|
|
86
|
+
return 2 unless lt_dir
|
|
87
|
+
unless install.java?
|
|
88
|
+
warn "error: java not found. Install a JRE (on macOS: brew install openjdk), " \
|
|
89
|
+
"or set SE_JAVA to your java binary."
|
|
90
|
+
return 2
|
|
91
|
+
end
|
|
92
|
+
unless SimpleEnglish::LanguageTool.smoke(install)
|
|
93
|
+
warn "error: LanguageTool smoke test failed. The download may be corrupt. " \
|
|
94
|
+
"Delete #{lt_dir} and rerun `se setup`."
|
|
95
|
+
return 2
|
|
96
|
+
end
|
|
97
|
+
puts <<~SETUP
|
|
98
|
+
LanguageTool #{SimpleEnglish::LanguageTool::LT_VERSION} is at #{lt_dir}
|
|
99
|
+
|
|
100
|
+
se finds it there automatically. Nothing to export.
|
|
101
|
+
SETUP
|
|
102
|
+
0
|
|
103
|
+
end
|
|
104
|
+
|
|
105
|
+
# "-" stays as-is for stdin. Directories expand to all lintable files.
|
|
106
|
+
# Glob output keeps a "./" prefix when the argument is ".". Strip it so
|
|
107
|
+
# paths and config ignore globs always see the same form.
|
|
108
|
+
def self.expand_paths(argv)
|
|
109
|
+
extensions = (SimpleEnglish::Extractor::EXTENSION_LANGUAGES.keys.map { |e| e.delete_prefix(".") } + ["md"]).uniq.join(",")
|
|
110
|
+
argv.flat_map do |path|
|
|
111
|
+
if File.directory?(path)
|
|
112
|
+
Dir.glob(File.join(path, "**/*.{#{extensions}}")).sort
|
|
113
|
+
else
|
|
114
|
+
path
|
|
115
|
+
end
|
|
116
|
+
end.map { |path| path.delete_prefix("./") }
|
|
117
|
+
end
|
|
118
|
+
|
|
119
|
+
def self.report(results, format)
|
|
120
|
+
require "json"
|
|
121
|
+
case format
|
|
122
|
+
when "json"
|
|
123
|
+
puts JSON.pretty_generate(results.map do |path, finding|
|
|
124
|
+
{"path" => path, "line" => finding.line, "column" => finding.column,
|
|
125
|
+
"rule" => finding.rule, "message" => finding.message}
|
|
126
|
+
end)
|
|
127
|
+
when "sarif"
|
|
128
|
+
puts JSON.pretty_generate(
|
|
129
|
+
"version" => "2.1.0",
|
|
130
|
+
"$schema" => "https://json.schemastore.org/sarif-2.1.0.json",
|
|
131
|
+
"runs" => [{
|
|
132
|
+
"tool" => {"driver" => {"name" => "se"}},
|
|
133
|
+
"results" => results.map do |path, finding|
|
|
134
|
+
region = {"startLine" => finding.line}
|
|
135
|
+
region["startColumn"] = finding.column if finding.column
|
|
136
|
+
{"ruleId" => finding.rule, "level" => "error",
|
|
137
|
+
"message" => {"text" => finding.message},
|
|
138
|
+
"locations" => [{"physicalLocation" => {
|
|
139
|
+
"artifactLocation" => {"uri" => path},
|
|
140
|
+
"region" => region
|
|
141
|
+
}}]}
|
|
142
|
+
end
|
|
143
|
+
}]
|
|
144
|
+
)
|
|
145
|
+
else
|
|
146
|
+
results.each do |path, finding|
|
|
147
|
+
puts "#{path}:#{finding.line}: [#{finding.rule}] #{finding.message}"
|
|
148
|
+
end
|
|
149
|
+
end
|
|
150
|
+
end
|
|
151
|
+
end
|
|
152
|
+
end
|
|
@@ -0,0 +1,133 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
# HTTP client for the LanguageTool server and the se daemon.
|
|
4
|
+
|
|
5
|
+
require "json"
|
|
6
|
+
require "net/http"
|
|
7
|
+
require "rbconfig"
|
|
8
|
+
require "uri"
|
|
9
|
+
|
|
10
|
+
require_relative "plain_text"
|
|
11
|
+
require_relative "languagetool"
|
|
12
|
+
|
|
13
|
+
module SimpleEnglish
|
|
14
|
+
module Client
|
|
15
|
+
module_function
|
|
16
|
+
|
|
17
|
+
# LanguageTool reports offsets in Java UTF-16 code units.
|
|
18
|
+
# Astral characters (emoji) count as two. Count them so a
|
|
19
|
+
# newline comparison cannot drift past a boundary. A match
|
|
20
|
+
# starting at the newline itself belongs to the next line.
|
|
21
|
+
def offset_to_line(text, offset)
|
|
22
|
+
line = 1
|
|
23
|
+
units = 0
|
|
24
|
+
text.each_char do |char|
|
|
25
|
+
line += 1 if char == "\n" && units <= offset
|
|
26
|
+
return line if units >= offset
|
|
27
|
+
units += (char.ord > 0xFFFF) ? 2 : 1
|
|
28
|
+
end
|
|
29
|
+
line
|
|
30
|
+
end
|
|
31
|
+
|
|
32
|
+
DEFAULT_PORT = 8181
|
|
33
|
+
REQUEST_TIMEOUT = 30
|
|
34
|
+
|
|
35
|
+
def url
|
|
36
|
+
ENV.fetch("SE_SERVER_URL") { "http://localhost:#{DEFAULT_PORT}" }
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
def post(uri, params, read_timeout: REQUEST_TIMEOUT)
|
|
40
|
+
Net::HTTP.start(uri.host, uri.port, open_timeout: 5,
|
|
41
|
+
read_timeout: read_timeout) do |http|
|
|
42
|
+
request = Net::HTTP::Post.new(uri.request_uri)
|
|
43
|
+
request.set_form_data(params)
|
|
44
|
+
http.request(request)
|
|
45
|
+
end
|
|
46
|
+
end
|
|
47
|
+
|
|
48
|
+
private_class_method :post
|
|
49
|
+
|
|
50
|
+
# Pattern rules. `payload` is Markdown-stripped text (String) or an
|
|
51
|
+
# AnnotatedText::Result (code comments): one payload interface,
|
|
52
|
+
# #lt_params and #locate, either side of the daemon's LT request.
|
|
53
|
+
def check(payload, base_url: url)
|
|
54
|
+
payload = to_payload(payload)
|
|
55
|
+
params = {"language" => "en",
|
|
56
|
+
"enabledRules" => SimpleEnglish::LanguageTool.rule_ids.join(","),
|
|
57
|
+
"enabledOnly" => "true"}.merge(payload.lt_params)
|
|
58
|
+
response = post(URI("#{base_url}/v2/check"), params)
|
|
59
|
+
parse_matches(JSON.parse(response.body).fetch("matches"), payload)
|
|
60
|
+
end
|
|
61
|
+
|
|
62
|
+
def parse_matches(matches, payload)
|
|
63
|
+
payload = to_payload(payload)
|
|
64
|
+
matches.map do |match|
|
|
65
|
+
line, column = payload.locate(match.fetch("offset"))
|
|
66
|
+
Finding.new(line: line, column: column,
|
|
67
|
+
rule: match.fetch("rule").fetch("id"), message: match.fetch("message"))
|
|
68
|
+
end
|
|
69
|
+
end
|
|
70
|
+
|
|
71
|
+
def to_payload(payload)
|
|
72
|
+
payload.is_a?(String) ? PlainText.new(payload) : payload
|
|
73
|
+
end
|
|
74
|
+
|
|
75
|
+
private_class_method :to_payload
|
|
76
|
+
|
|
77
|
+
# Full lint via the se daemon. Raw Markdown in, or code
|
|
78
|
+
# source with a language for the comment pipeline. nil when
|
|
79
|
+
# unreachable or the response is unusable, so callers can fall
|
|
80
|
+
# back without seeing a stack trace.
|
|
81
|
+
def lint(text, base_url: url, language: nil)
|
|
82
|
+
params = {"text" => text}
|
|
83
|
+
params["language"] = language if language
|
|
84
|
+
response = post(URI("#{base_url}/lint"), params)
|
|
85
|
+
return nil unless response.is_a?(Net::HTTPSuccess)
|
|
86
|
+
body = JSON.parse(response.body)
|
|
87
|
+
return nil unless body.is_a?(Array)
|
|
88
|
+
body.map do |hash|
|
|
89
|
+
Finding.new(line: hash.fetch("line"), column: hash["column"],
|
|
90
|
+
rule: hash.fetch("rule"), message: hash.fetch("message"))
|
|
91
|
+
end
|
|
92
|
+
rescue SystemCallError, SocketError, Timeout::Error,
|
|
93
|
+
JSON::ParserError, TypeError
|
|
94
|
+
nil
|
|
95
|
+
end
|
|
96
|
+
|
|
97
|
+
def up?(base_url: url)
|
|
98
|
+
uri = URI(base_url)
|
|
99
|
+
Net::HTTP.start(uri.host, uri.port, open_timeout: 1,
|
|
100
|
+
read_timeout: 2) { |http| http.head("/") }
|
|
101
|
+
true
|
|
102
|
+
rescue Errno::ECONNREFUSED, SocketError, Timeout::Error
|
|
103
|
+
false
|
|
104
|
+
end
|
|
105
|
+
|
|
106
|
+
# True when the daemon answers. Starts it when we own the default
|
|
107
|
+
# URL. A custom SE_SERVER_URL belongs to someone else, so
|
|
108
|
+
# never spawn against it. It returns false when unusable. The
|
|
109
|
+
# caller owns the exit status. Diagnostics go to stderr here, where the cause
|
|
110
|
+
# is known.
|
|
111
|
+
def ensure_up(install: SimpleEnglish::Install.from_env)
|
|
112
|
+
unless File.exist?(install.server_jar)
|
|
113
|
+
warn install.setup_error
|
|
114
|
+
return false
|
|
115
|
+
end
|
|
116
|
+
return true if up?
|
|
117
|
+
if ENV["SE_SERVER_URL"]
|
|
118
|
+
warn "error: SE_SERVER_URL is set but #{url} does not answer."
|
|
119
|
+
return false
|
|
120
|
+
end
|
|
121
|
+
warn "se: daemon not running; starting it (first lint takes ~15s)..."
|
|
122
|
+
bin = File.expand_path("../../bin/se", __dir__)
|
|
123
|
+
Process.spawn(RbConfig.ruby, bin, "serve", out: File::NULL, err: File::NULL)
|
|
124
|
+
deadline = Time.now + 90
|
|
125
|
+
until Time.now > deadline
|
|
126
|
+
return true if up?
|
|
127
|
+
sleep 0.5
|
|
128
|
+
end
|
|
129
|
+
warn "error: se daemon did not come up. Run `se serve` and read its output."
|
|
130
|
+
false
|
|
131
|
+
end
|
|
132
|
+
end
|
|
133
|
+
end
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
# .simple-english.yml: ignore (path globs, matched against the paths given
|
|
4
|
+
# on the command line) and disabled-rules. Loaded from the CWD.
|
|
5
|
+
|
|
6
|
+
module SimpleEnglish
|
|
7
|
+
module Config
|
|
8
|
+
ConfigError = Class.new(StandardError)
|
|
9
|
+
DEFAULT = {ignore: [], disabled_rules: []}.freeze
|
|
10
|
+
|
|
11
|
+
module_function
|
|
12
|
+
|
|
13
|
+
def load(dir = Dir.pwd)
|
|
14
|
+
file = File.join(dir, ".simple-english.yml")
|
|
15
|
+
return DEFAULT unless File.exist?(file)
|
|
16
|
+
require "yaml"
|
|
17
|
+
data = YAML.safe_load_file(file) || {}
|
|
18
|
+
{ignore: Array(data["ignore"]),
|
|
19
|
+
disabled_rules: Array(data["disabled-rules"])}
|
|
20
|
+
rescue Psych::SyntaxError => e
|
|
21
|
+
raise ConfigError, ".simple-english.yml: #{e.message}"
|
|
22
|
+
end
|
|
23
|
+
|
|
24
|
+
def ignore?(config, path)
|
|
25
|
+
config[:ignore].any? { |pattern| matches?(pattern, path) }
|
|
26
|
+
end
|
|
27
|
+
|
|
28
|
+
# File.fnmatch has no globstar: `**` never crosses directories.
|
|
29
|
+
# Translate instead: `**` as a whole segment is any depth, `*` and
|
|
30
|
+
# `?` stay within one segment.
|
|
31
|
+
def matches?(pattern, path)
|
|
32
|
+
glob_to_regex(pattern).match?(path)
|
|
33
|
+
end
|
|
34
|
+
|
|
35
|
+
def glob_to_regex(pattern)
|
|
36
|
+
Regexp.new("\\A" + pattern.split("/").map do |segment|
|
|
37
|
+
if segment == "**"
|
|
38
|
+
"(?:[^/]+/)*[^/]*"
|
|
39
|
+
else
|
|
40
|
+
Regexp.escape(segment).gsub("\\*", "[^/]*").gsub("\\?", "[^/]")
|
|
41
|
+
end
|
|
42
|
+
end.join("/") + "\\z")
|
|
43
|
+
end
|
|
44
|
+
|
|
45
|
+
def filter(config, path, findings)
|
|
46
|
+
return [] if ignore?(config, path)
|
|
47
|
+
disabled = config[:disabled_rules]
|
|
48
|
+
findings.reject { |finding| disabled.include?(finding.rule) }
|
|
49
|
+
end
|
|
50
|
+
end
|
|
51
|
+
end
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
# Counting rules: sentence and paragraph length. Pure text analysis.
|
|
4
|
+
|
|
5
|
+
module SimpleEnglish
|
|
6
|
+
module Counts
|
|
7
|
+
PROCEDURAL_WORD_LIMIT = 20
|
|
8
|
+
DESCRIPTIVE_WORD_LIMIT = 25
|
|
9
|
+
PARAGRAPH_SENTENCE_LIMIT = 6
|
|
10
|
+
|
|
11
|
+
module_function
|
|
12
|
+
|
|
13
|
+
# Takes Markdown-stripped text, returns findings for long
|
|
14
|
+
# sentences and long paragraphs.
|
|
15
|
+
def check(text)
|
|
16
|
+
Markdown.paragraphs(text).flat_map { |paragraph| findings_for(paragraph) }
|
|
17
|
+
end
|
|
18
|
+
|
|
19
|
+
def findings_for(paragraph)
|
|
20
|
+
sentences = Markdown.sentences_of(paragraph.lines.join(" "))
|
|
21
|
+
limit = paragraph.procedural ? PROCEDURAL_WORD_LIMIT : DESCRIPTIVE_WORD_LIMIT
|
|
22
|
+
|
|
23
|
+
findings = sentences
|
|
24
|
+
.select { |sentence| sentence.split.size > limit }
|
|
25
|
+
.map do |_sentence|
|
|
26
|
+
Finding.new(line: paragraph.start_line, column: nil, rule: "SE_SENTENCE_TOO_LONG",
|
|
27
|
+
message: "Sentence has more than #{limit} words. Split it.")
|
|
28
|
+
end
|
|
29
|
+
|
|
30
|
+
if !paragraph.procedural && sentences.size > PARAGRAPH_SENTENCE_LIMIT
|
|
31
|
+
findings << Finding.new(line: paragraph.start_line, column: nil, rule: "SE_PARAGRAPH_TOO_LONG",
|
|
32
|
+
message: "Paragraph has more than #{PARAGRAPH_SENTENCE_LIMIT} " \
|
|
33
|
+
"sentences. Give one topic six sentences at most.")
|
|
34
|
+
end
|
|
35
|
+
findings
|
|
36
|
+
end
|
|
37
|
+
|
|
38
|
+
private_class_method :findings_for
|
|
39
|
+
end
|
|
40
|
+
end
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
# The complete lint engine, daemon-side. Never call SimpleEnglish.lint_text
|
|
4
|
+
# here: that call probes the daemon and recurses into the server.
|
|
5
|
+
|
|
6
|
+
require "json"
|
|
7
|
+
|
|
8
|
+
module SimpleEnglish
|
|
9
|
+
module Engine
|
|
10
|
+
module_function
|
|
11
|
+
|
|
12
|
+
def lint(text, base_url:, language: nil)
|
|
13
|
+
if language
|
|
14
|
+
spans = SimpleEnglish::Extractor.comment_spans(text, language)
|
|
15
|
+
spans.empty? ? [] :
|
|
16
|
+
SimpleEnglish::Client.check(
|
|
17
|
+
SimpleEnglish::AnnotatedText.build(text, spans), base_url: base_url
|
|
18
|
+
)
|
|
19
|
+
else
|
|
20
|
+
stripped = SimpleEnglish::Markdown.strip(text)
|
|
21
|
+
(SimpleEnglish::Counts.check(stripped) +
|
|
22
|
+
SimpleEnglish::Client.check(stripped, base_url: base_url))
|
|
23
|
+
.sort_by { |finding| [finding.line, finding.rule] }
|
|
24
|
+
end
|
|
25
|
+
end
|
|
26
|
+
|
|
27
|
+
def lint_json(text, base_url:, language: nil)
|
|
28
|
+
JSON.generate(lint(text, base_url: base_url, language: language).map(&:to_h))
|
|
29
|
+
end
|
|
30
|
+
end
|
|
31
|
+
end
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
# Tree-sitter comment extraction. One walk, no per-language code:
|
|
4
|
+
# every grammar marks comments with a kind ending in "comment".
|
|
5
|
+
|
|
6
|
+
require_relative "span"
|
|
7
|
+
|
|
8
|
+
module SimpleEnglish
|
|
9
|
+
module Extractor
|
|
10
|
+
EXTENSION_LANGUAGES = {
|
|
11
|
+
".py" => "python",
|
|
12
|
+
".rb" => "ruby",
|
|
13
|
+
".js" => "javascript",
|
|
14
|
+
".mjs" => "javascript",
|
|
15
|
+
".ts" => "typescript",
|
|
16
|
+
".yaml" => "yaml",
|
|
17
|
+
".yml" => "yaml",
|
|
18
|
+
".go" => "go",
|
|
19
|
+
".rs" => "rust",
|
|
20
|
+
".java" => "java",
|
|
21
|
+
".sh" => "bash",
|
|
22
|
+
".kt" => "kotlin",
|
|
23
|
+
".cs" => "csharp"
|
|
24
|
+
}.freeze
|
|
25
|
+
|
|
26
|
+
module_function
|
|
27
|
+
|
|
28
|
+
# nil for files we do not lint.
|
|
29
|
+
def language_for(path)
|
|
30
|
+
EXTENSION_LANGUAGES[File.extname(path)]
|
|
31
|
+
end
|
|
32
|
+
|
|
33
|
+
def comment_spans(source, language)
|
|
34
|
+
require "tree_sitter_language_pack"
|
|
35
|
+
root = TreeSitterLanguagePack.get_parser(language).parse(source).root_node
|
|
36
|
+
spans = []
|
|
37
|
+
stack = [root]
|
|
38
|
+
until stack.empty?
|
|
39
|
+
node = stack.pop
|
|
40
|
+
if comment?(node)
|
|
41
|
+
spans << Span.new(text: source.byteslice(node.start_byte...node.end_byte),
|
|
42
|
+
start_byte: node.start_byte, end_byte: node.end_byte)
|
|
43
|
+
end
|
|
44
|
+
(node.child_count - 1).downto(0) { |i| stack.push(node.child(i)) }
|
|
45
|
+
end
|
|
46
|
+
spans.sort_by!(&:start_byte)
|
|
47
|
+
end
|
|
48
|
+
|
|
49
|
+
def comment?(node)
|
|
50
|
+
node.kind.end_with?("comment") && !node.kind.start_with?("non")
|
|
51
|
+
end
|
|
52
|
+
|
|
53
|
+
private_class_method :comment?
|
|
54
|
+
end
|
|
55
|
+
end
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
# Minimal HTTP/1.1 wire framing for the daemon's outer server. Serving
|
|
4
|
+
# one request per connection (Connection: close) keeps this simple: no
|
|
5
|
+
# chunked bodies, no pipelining.
|
|
6
|
+
|
|
7
|
+
require "json"
|
|
8
|
+
|
|
9
|
+
module SimpleEnglish
|
|
10
|
+
module HTTP
|
|
11
|
+
module_function
|
|
12
|
+
|
|
13
|
+
# Reads one HTTP request from io: request line, headers, then
|
|
14
|
+
# Content-Length bytes of body. nil when malformed or the peer
|
|
15
|
+
# hung up.
|
|
16
|
+
def read_request(io)
|
|
17
|
+
request_line = io.gets
|
|
18
|
+
parts = request_line ? request_line.split(" ") : []
|
|
19
|
+
return nil if parts.length < 2
|
|
20
|
+
content_length = 0
|
|
21
|
+
while (line = io.gets)
|
|
22
|
+
break if line == "\r\n" || line == "\n"
|
|
23
|
+
name, value = line.split(":", 2)
|
|
24
|
+
next unless value
|
|
25
|
+
content_length = value.strip.to_i if name.casecmp?("content-length")
|
|
26
|
+
end
|
|
27
|
+
return nil if content_length.negative?
|
|
28
|
+
body = content_length.zero? ? +"" : io.read(content_length)
|
|
29
|
+
{method: parts[0], path: parts[1], body: body}
|
|
30
|
+
end
|
|
31
|
+
|
|
32
|
+
# Writes a minimal HTTP/1.1 response and closes the logical
|
|
33
|
+
# connection (Connection: close): status line, always-JSON headers,
|
|
34
|
+
# Content-Length, then the body.
|
|
35
|
+
def write_response(io, status:, body:, content_type: "application/json")
|
|
36
|
+
reason = {200 => "OK", 404 => "Not Found",
|
|
37
|
+
500 => "Internal Server Error"}.fetch(status, "Unknown")
|
|
38
|
+
io.write("HTTP/1.1 #{status} #{reason}\r\n" \
|
|
39
|
+
"Content-Type: #{content_type}\r\n" \
|
|
40
|
+
"Content-Length: #{body.bytesize}\r\n" \
|
|
41
|
+
"Connection: close\r\n\r\n")
|
|
42
|
+
io.write(body)
|
|
43
|
+
end
|
|
44
|
+
|
|
45
|
+
# One thread per request: lint can take seconds, so a slow request
|
|
46
|
+
# must not block readiness probes or other clients.
|
|
47
|
+
def handle_client(client, port:)
|
|
48
|
+
request = read_request(client)
|
|
49
|
+
if request.nil?
|
|
50
|
+
# Malformed request or immediate hangup: nothing to answer.
|
|
51
|
+
elsif request[:method] == "HEAD"
|
|
52
|
+
# Client.up? does http.head("/") and treats any response as up.
|
|
53
|
+
write_response(client, status: 200, body: "")
|
|
54
|
+
elsif request[:method] == "POST" && request[:path] == "/lint"
|
|
55
|
+
begin
|
|
56
|
+
# Ruling 2026-09-24: Client.lint posts form-encoded data, so the
|
|
57
|
+
# handler decodes it. A raw body lints "text=Don%27t...".
|
|
58
|
+
# The optional language switches to the code-comment pipeline.
|
|
59
|
+
params = URI.decode_www_form(request[:body]).to_h
|
|
60
|
+
write_response(client, status: 200,
|
|
61
|
+
body: SimpleEnglish::Engine.lint_json(params["text"],
|
|
62
|
+
base_url: "http://localhost:#{port + 1}",
|
|
63
|
+
language: params["language"]))
|
|
64
|
+
rescue => e
|
|
65
|
+
write_response(client, status: 500,
|
|
66
|
+
body: JSON.generate({"error" => e.message}))
|
|
67
|
+
end
|
|
68
|
+
else
|
|
69
|
+
write_response(client, status: 404,
|
|
70
|
+
body: JSON.generate({"error" => "not found"}))
|
|
71
|
+
end
|
|
72
|
+
ensure
|
|
73
|
+
client.close
|
|
74
|
+
end
|
|
75
|
+
end
|
|
76
|
+
end
|