simple_english 0.2.0 → 0.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +15 -9
- data/lib/simple_english/cli.rb +17 -2
- data/lib/simple_english/client.rb +34 -33
- data/lib/simple_english/extractor.rb +6 -1
- data/lib/simple_english/finding.rb +4 -3
- data/lib/simple_english/markdown.rb +28 -10
- data/lib/simple_english/plain_text.rb +1 -2
- data/lib/simple_english/version.rb +1 -1
- data/rules/simple-english.xml +5 -1
- metadata +6 -3
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: 5c1efe5da86f44ce1618aea7d2c49f3e5dd0aec897a8f103ace5276be73d9ef4
|
|
4
|
+
data.tar.gz: cd8597b07d037a0631d16ea11377144fb9b93d6f6e604a316027d5c43a72d50d
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: 838420d06071034d7e6d48c074fc0d5893b6443e6f1dd387247e86239f74b620ff7e4adecd69e4b37a3d58e98ebb193bbe53a113dc96759a075cb08fdd8247c7
|
|
7
|
+
data.tar.gz: 7fcc415dcf6cc5a01ba01807118ca6e3ea8ff7f7d4573960f5b31fe7971c3612559e15d5f7ee0c122363cdf2b620b8ede79ed7b0e0e193dd188a0446483045fc
|
data/README.md
CHANGED
|
@@ -11,16 +11,18 @@ the slop it leaves behind. Every finding says what to write
|
|
|
11
11
|
instead.
|
|
12
12
|
|
|
13
13
|
```console
|
|
14
|
-
$ printf
|
|
14
|
+
$ printf "The config was written by setup — don't edit it; the daemon caches rules, making the first lint slow." > note.md
|
|
15
15
|
$ se note.md
|
|
16
|
-
note.md:1:
|
|
17
|
-
note.md:1:
|
|
18
|
-
note.md:1:
|
|
16
|
+
note.md:1:12-26: [SE_ACTIVE_VOICE] "was written by" - Use the active voice. Say who does the action.
|
|
17
|
+
note.md:1:73-81: [SE_ING_AFTER_COMMA] ", making" - Start a new sentence instead of the -ing phrase.
|
|
18
|
+
note.md:1:37-40: [SE_NO_CONTRACTIONS] "n't" - Write the words in full. No contractions.
|
|
19
|
+
note.md:1:33-34: [SE_NO_EMDASH] "—" - Write two sentences, or use a comma.
|
|
20
|
+
note.md:1:48-49: [SE_NO_SEMICOLON] ";" - Write two sentences, or name the relation.
|
|
19
21
|
```
|
|
20
22
|
|
|
21
23
|
Markdown prose plus code comments in Python, Ruby, JavaScript,
|
|
22
|
-
TypeScript, Go, Rust, Java, C#, Kotlin, bash, and YAML. Output as
|
|
23
|
-
text, JSON, or SARIF.
|
|
24
|
+
TypeScript, Go, Rust, Java, C#, C++, Kotlin, bash, and YAML. Output as
|
|
25
|
+
plain text, JSON, or SARIF.
|
|
24
26
|
|
|
25
27
|
## The rules
|
|
26
28
|
|
|
@@ -31,7 +33,7 @@ text, JSON, or SARIF.
|
|
|
31
33
|
- **Contractions:** write every word in full.
|
|
32
34
|
- **Sentence shape:** condition before command, no `-ing` phrase after a comma.
|
|
33
35
|
- **Word choice:** about 50 substitution rules, from `leverage` to `in conclusion`. `make sure that` keeps its "that".
|
|
34
|
-
- **Code comments:** same pattern rules, with line and column.
|
|
36
|
+
- **Code comments:** same pattern rules, with line and column range.
|
|
35
37
|
- **Counts (Markdown only):** 20 words per sentence in list items, 25 in paragraphs, six sentences per paragraph at most.
|
|
36
38
|
|
|
37
39
|
The full list, with a wrong and a right example for each rule:
|
|
@@ -93,7 +95,11 @@ git diff --name-only --diff-filter=ACM main | xargs se
|
|
|
93
95
|
|
|
94
96
|
### Outputs
|
|
95
97
|
|
|
96
|
-
|
|
98
|
+
Pattern findings print an exclusive column range as
|
|
99
|
+
`file:line:start-end: [RULE_ID] message`. A range that crosses lines ends
|
|
100
|
+
with `end-line:end-column`. Columns count UTF-16 code units. Counting findings
|
|
101
|
+
identify only the paragraph's first line. JSON and SARIF output expose the same
|
|
102
|
+
source ranges:
|
|
97
103
|
|
|
98
104
|
```bash
|
|
99
105
|
se --format json docs/
|
|
@@ -180,7 +186,7 @@ Any tool or language can lint through its HTTP API:
|
|
|
180
186
|
|
|
181
187
|
```bash
|
|
182
188
|
curl -d "text=Don't do this." http://localhost:8181/lint
|
|
183
|
-
# [{"line":1,"column":
|
|
189
|
+
# [{"line":1,"column":3,"end_line":1,"end_column":6,"rule":"SE_NO_CONTRACTIONS","message":"..."}]
|
|
184
190
|
```
|
|
185
191
|
|
|
186
192
|
The full wire format: [docs/DAEMON.md](docs/DAEMON.md).
|
data/lib/simple_english/cli.rb
CHANGED
|
@@ -122,6 +122,7 @@ module SimpleEnglish
|
|
|
122
122
|
when "json"
|
|
123
123
|
puts JSON.pretty_generate(results.map do |path, finding|
|
|
124
124
|
{"path" => path, "line" => finding.line, "column" => finding.column,
|
|
125
|
+
"end_line" => finding.end_line, "end_column" => finding.end_column,
|
|
125
126
|
"rule" => finding.rule, "message" => finding.message}
|
|
126
127
|
end)
|
|
127
128
|
when "sarif"
|
|
@@ -129,10 +130,17 @@ module SimpleEnglish
|
|
|
129
130
|
"version" => "2.1.0",
|
|
130
131
|
"$schema" => "https://json.schemastore.org/sarif-2.1.0.json",
|
|
131
132
|
"runs" => [{
|
|
133
|
+
"columnKind" => "utf16CodeUnits",
|
|
132
134
|
"tool" => {"driver" => {"name" => "se"}},
|
|
133
135
|
"results" => results.map do |path, finding|
|
|
134
136
|
region = {"startLine" => finding.line}
|
|
135
|
-
|
|
137
|
+
if finding.column
|
|
138
|
+
region["startColumn"] = finding.column
|
|
139
|
+
if finding.end_line && finding.end_column
|
|
140
|
+
region["endLine"] = finding.end_line
|
|
141
|
+
region["endColumn"] = finding.end_column
|
|
142
|
+
end
|
|
143
|
+
end
|
|
136
144
|
{"ruleId" => finding.rule, "level" => "error",
|
|
137
145
|
"message" => {"text" => finding.message},
|
|
138
146
|
"locations" => [{"physicalLocation" => {
|
|
@@ -145,7 +153,14 @@ module SimpleEnglish
|
|
|
145
153
|
else
|
|
146
154
|
results.each do |path, finding|
|
|
147
155
|
location = "#{path}:#{finding.line}"
|
|
148
|
-
|
|
156
|
+
if finding.column
|
|
157
|
+
location += ":#{finding.column}"
|
|
158
|
+
if finding.end_line && finding.end_column
|
|
159
|
+
finish = (finding.end_line == finding.line) ? finding.end_column :
|
|
160
|
+
"#{finding.end_line}:#{finding.end_column}"
|
|
161
|
+
location += "-#{finish}"
|
|
162
|
+
end
|
|
163
|
+
end
|
|
149
164
|
puts "#{location}: [#{finding.rule}] #{finding.message}"
|
|
150
165
|
end
|
|
151
166
|
end
|
|
@@ -14,39 +14,23 @@ module SimpleEnglish
|
|
|
14
14
|
module Client
|
|
15
15
|
module_function
|
|
16
16
|
|
|
17
|
-
# LanguageTool reports offsets in Java UTF-16 code units.
|
|
18
|
-
#
|
|
19
|
-
|
|
20
|
-
# starting at the newline itself belongs to the next line.
|
|
21
|
-
def offset_to_line(text, offset)
|
|
17
|
+
# LanguageTool reports offsets and lengths in Java UTF-16 code units.
|
|
18
|
+
# Return a 1-based line and UTF-16 column for its 0-based offset.
|
|
19
|
+
def offset_to_position(text, offset)
|
|
22
20
|
line = 1
|
|
21
|
+
column = 1
|
|
23
22
|
units = 0
|
|
24
23
|
text.each_char do |char|
|
|
25
|
-
line
|
|
26
|
-
return line if units >= offset
|
|
24
|
+
return [line, column] if units >= offset
|
|
27
25
|
units += (char.ord > 0xFFFF) ? 2 : 1
|
|
28
|
-
end
|
|
29
|
-
line
|
|
30
|
-
end
|
|
31
|
-
|
|
32
|
-
# The 1-based column of a UTF-16 offset within its line. Astral
|
|
33
|
-
# characters count as two units, like the match offset. A match
|
|
34
|
-
# starting at the newline itself is column one of the next line.
|
|
35
|
-
def offset_to_column(text, offset)
|
|
36
|
-
column = 0
|
|
37
|
-
units = 0
|
|
38
|
-
text.each_char do |char|
|
|
39
26
|
if char == "\n"
|
|
40
|
-
|
|
41
|
-
column =
|
|
42
|
-
|
|
43
|
-
|
|
27
|
+
line += 1
|
|
28
|
+
column = 1
|
|
29
|
+
else
|
|
30
|
+
column += (char.ord > 0xFFFF) ? 2 : 1
|
|
44
31
|
end
|
|
45
|
-
return column + 1 if units >= offset
|
|
46
|
-
units += (char.ord > 0xFFFF) ? 2 : 1
|
|
47
|
-
column += 1
|
|
48
32
|
end
|
|
49
|
-
column
|
|
33
|
+
[line, column]
|
|
50
34
|
end
|
|
51
35
|
|
|
52
36
|
DEFAULT_PORT = 8181
|
|
@@ -82,29 +66,45 @@ module SimpleEnglish
|
|
|
82
66
|
def parse_matches(matches, payload)
|
|
83
67
|
payload = to_payload(payload)
|
|
84
68
|
matches.map do |match|
|
|
85
|
-
|
|
69
|
+
offset = match.fetch("offset")
|
|
70
|
+
line, column = payload.locate(offset)
|
|
71
|
+
end_line, end_column = payload.locate(offset + match.fetch("length"))
|
|
86
72
|
Finding.new(line: line, column: column,
|
|
73
|
+
end_line: end_line, end_column: end_column,
|
|
87
74
|
rule: match.fetch("rule").fetch("id"),
|
|
88
75
|
message: with_context(match.fetch("message"), match))
|
|
89
76
|
end
|
|
90
77
|
end
|
|
91
78
|
|
|
92
79
|
# Prefix the offending text, so a finding says what to change,
|
|
93
|
-
# not only how.
|
|
94
|
-
# Context offsets are Java UTF-16 code units, like the match
|
|
95
|
-
# offset.
|
|
80
|
+
# not only how. Context positions use Java UTF-16 code units.
|
|
96
81
|
def with_context(message, match)
|
|
97
82
|
context = match["context"] or return message
|
|
98
|
-
matched = context.fetch("text", "")
|
|
99
|
-
context.fetch("offset", 0).to_i, context.fetch("length", 0).to_i
|
|
83
|
+
matched = utf16_slice(context.fetch("text", ""),
|
|
84
|
+
context.fetch("offset", 0).to_i, context.fetch("length", 0).to_i)
|
|
100
85
|
matched.empty? ? message : "\"#{matched}\" - #{message}"
|
|
101
86
|
end
|
|
102
87
|
|
|
88
|
+
def utf16_slice(text, offset, length)
|
|
89
|
+
first = utf16_index(text, offset)
|
|
90
|
+
last = utf16_index(text, offset + length)
|
|
91
|
+
text[first...last]
|
|
92
|
+
end
|
|
93
|
+
|
|
94
|
+
def utf16_index(text, offset)
|
|
95
|
+
units = 0
|
|
96
|
+
text.each_char.with_index do |char, index|
|
|
97
|
+
return index if units >= offset
|
|
98
|
+
units += (char.ord > 0xFFFF) ? 2 : 1
|
|
99
|
+
end
|
|
100
|
+
text.length
|
|
101
|
+
end
|
|
102
|
+
|
|
103
103
|
def to_payload(payload)
|
|
104
104
|
payload.is_a?(String) ? PlainText.new(payload) : payload
|
|
105
105
|
end
|
|
106
106
|
|
|
107
|
-
private_class_method :to_payload, :with_context
|
|
107
|
+
private_class_method :to_payload, :with_context, :utf16_slice, :utf16_index
|
|
108
108
|
|
|
109
109
|
# Full lint via the se daemon. Raw Markdown in, or code
|
|
110
110
|
# source with a language for the comment pipeline. nil when
|
|
@@ -119,6 +119,7 @@ module SimpleEnglish
|
|
|
119
119
|
return nil unless body.is_a?(Array)
|
|
120
120
|
body.map do |hash|
|
|
121
121
|
Finding.new(line: hash.fetch("line"), column: hash["column"],
|
|
122
|
+
end_line: hash["end_line"], end_column: hash["end_column"],
|
|
122
123
|
rule: hash.fetch("rule"), message: hash.fetch("message"))
|
|
123
124
|
end
|
|
124
125
|
rescue SystemCallError, SocketError, Timeout::Error,
|
|
@@ -20,7 +20,12 @@ module SimpleEnglish
|
|
|
20
20
|
".java" => "java",
|
|
21
21
|
".sh" => "bash",
|
|
22
22
|
".kt" => "kotlin",
|
|
23
|
-
".cs" => "csharp"
|
|
23
|
+
".cs" => "csharp",
|
|
24
|
+
".cpp" => "cpp",
|
|
25
|
+
".cc" => "cpp",
|
|
26
|
+
".cxx" => "cpp",
|
|
27
|
+
".hpp" => "cpp",
|
|
28
|
+
".h" => "cpp"
|
|
24
29
|
}.freeze
|
|
25
30
|
|
|
26
31
|
module_function
|
|
@@ -1,8 +1,9 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
|
-
# One lint result.
|
|
4
|
-
#
|
|
3
|
+
# One lint result. Positions are 1-based. The end position is exclusive.
|
|
4
|
+
# Counting findings have no columns or end positions. Findings from an older
|
|
5
|
+
# daemon can have a start column without an end position.
|
|
5
6
|
|
|
6
7
|
module SimpleEnglish
|
|
7
|
-
Finding = Struct.new(:line, :column, :rule, :message)
|
|
8
|
+
Finding = Struct.new(:line, :column, :end_line, :end_column, :rule, :message)
|
|
8
9
|
end
|
|
@@ -21,21 +21,39 @@ module SimpleEnglish
|
|
|
21
21
|
module_function
|
|
22
22
|
|
|
23
23
|
# Blank out code and other non-prose blocks and neutralize inline
|
|
24
|
-
# code and heading markers.
|
|
25
|
-
# source.
|
|
24
|
+
# code and heading markers. Line count and UTF-16 widths stay identical
|
|
25
|
+
# to the source so LanguageTool offsets remain source positions.
|
|
26
26
|
def strip(text)
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
27
|
+
blank = blank_rows(text).to_h { |row| [row, true] }
|
|
28
|
+
text.lines.each_with_index.map do |line, row|
|
|
29
|
+
ending = if line.end_with?("\r\n")
|
|
30
|
+
"\r\n"
|
|
31
|
+
elsif line.end_with?("\n")
|
|
32
|
+
"\n"
|
|
33
|
+
else
|
|
34
|
+
""
|
|
35
|
+
end
|
|
36
|
+
body = line.delete_suffix(ending)
|
|
37
|
+
(blank[row] ? spaces_for(body) : strip_line(body)) + ending
|
|
38
|
+
end.join
|
|
30
39
|
end
|
|
31
40
|
|
|
32
41
|
def strip_line(line)
|
|
33
|
-
# Width-preserving replacements
|
|
34
|
-
# the source
|
|
35
|
-
line
|
|
42
|
+
# Width-preserving replacements keep LanguageTool positions aligned
|
|
43
|
+
# with the source.
|
|
44
|
+
line
|
|
45
|
+
.gsub(/`[^`]*`/) { |code| "X" * utf16_length(code) }
|
|
36
46
|
.sub(/\A\#{1,6} /) { |marker| " " * marker.length }
|
|
37
47
|
end
|
|
38
48
|
|
|
49
|
+
def spaces_for(text)
|
|
50
|
+
" " * utf16_length(text)
|
|
51
|
+
end
|
|
52
|
+
|
|
53
|
+
def utf16_length(text)
|
|
54
|
+
text.each_char.sum { |char| (char.ord > 0xFFFF) ? 2 : 1 }
|
|
55
|
+
end
|
|
56
|
+
|
|
39
57
|
# A vertical list is not one paragraph: each list item is its own.
|
|
40
58
|
# Paragraph boundaries come from the tree-sitter Markdown grammar,
|
|
41
59
|
# so lazy continuations and interrupted lists match the spec
|
|
@@ -97,7 +115,7 @@ module SimpleEnglish
|
|
|
97
115
|
TreeSitterLanguagePack.get_parser("markdown").parse(text).root_node
|
|
98
116
|
end
|
|
99
117
|
|
|
100
|
-
private_class_method :strip_line, :
|
|
101
|
-
:inside_list_item?, :parse
|
|
118
|
+
private_class_method :strip_line, :spaces_for, :utf16_length, :node_rows,
|
|
119
|
+
:blank_rows, :each_node, :inside_list_item?, :parse
|
|
102
120
|
end
|
|
103
121
|
end
|
data/rules/simple-english.xml
CHANGED
|
@@ -56,10 +56,14 @@
|
|
|
56
56
|
</rule>
|
|
57
57
|
|
|
58
58
|
<rule id="SE_ING_AFTER_COMMA" name="No -ing after a comma">
|
|
59
|
-
<pattern><token>,</token><token postag="VBG"
|
|
59
|
+
<pattern><token>,</token><token postag="VBG"><exception scope="next" regexp="yes">[,)]|and|or</exception></token></pattern>
|
|
60
60
|
<message>Start a new sentence instead of the -ing phrase.</message>
|
|
61
61
|
<example type="incorrect">The tool runs<marker>, making</marker> it easy.</example>
|
|
62
62
|
<example type="correct">The tool runs. It is easy.</example>
|
|
63
|
+
<example type="correct">The walkthrough (accounts, VPC, secrets, troubleshooting) is in the guide.</example>
|
|
64
|
+
<example type="correct">The guide covers billing, troubleshooting, and support for each environment.</example>
|
|
65
|
+
<example type="correct">The guide covers billing, troubleshooting and support for each environment.</example>
|
|
66
|
+
<example type="correct">The guide covers billing, troubleshooting or support for each environment.</example>
|
|
63
67
|
</rule>
|
|
64
68
|
|
|
65
69
|
<rule id="SE_KEEP_THAT" name="make sure that">
|
metadata
CHANGED
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
--- !ruby/object:Gem::Specification
|
|
2
2
|
name: simple_english
|
|
3
3
|
version: !ruby/object:Gem::Version
|
|
4
|
-
version: 0.
|
|
4
|
+
version: 0.3.0
|
|
5
5
|
platform: ruby
|
|
6
6
|
authors:
|
|
7
7
|
- TonyCTHsu
|
|
8
8
|
autorequire:
|
|
9
9
|
bindir: bin
|
|
10
10
|
cert_chain: []
|
|
11
|
-
date: 2026-09-
|
|
11
|
+
date: 2026-09-29 00:00:00.000000000 Z
|
|
12
12
|
dependencies:
|
|
13
13
|
- !ruby/object:Gem::Dependency
|
|
14
14
|
name: tree_sitter_language_pack
|
|
@@ -159,7 +159,10 @@ files:
|
|
|
159
159
|
homepage: https://github.com/TonyCTHsu/simple-english
|
|
160
160
|
licenses:
|
|
161
161
|
- MIT
|
|
162
|
-
metadata:
|
|
162
|
+
metadata:
|
|
163
|
+
homepage_uri: https://github.com/TonyCTHsu/simple-english
|
|
164
|
+
source_code_uri: https://github.com/TonyCTHsu/simple-english
|
|
165
|
+
changelog_uri: https://github.com/TonyCTHsu/simple-english/blob/v0.3.0/CHANGELOG.md
|
|
163
166
|
post_install_message:
|
|
164
167
|
rdoc_options: []
|
|
165
168
|
require_paths:
|