sloplint 0.8.0 → 0.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +123 -1
- data/README.md +90 -5
- data/docs/SPEC.md +31 -6
- data/exe/sloplint +2 -1
- data/lib/sloplint/cli.rb +208 -40
- data/lib/sloplint/engine.rb +31 -4
- data/lib/sloplint/output.rb +9 -5
- data/lib/sloplint/rules.rb +127 -0
- data/lib/sloplint/split.rb +348 -0
- data/lib/sloplint/version.rb +1 -1
- metadata +2 -1
|
@@ -0,0 +1,348 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require_relative "rules"
|
|
4
|
+
require_relative "engine"
|
|
5
|
+
|
|
6
|
+
module Sloplint
|
|
7
|
+
# Splits a document into paragraphs and sentences, keeping each one's offset
|
|
8
|
+
# and length in the original text so a note can land on the file as written.
|
|
9
|
+
#
|
|
10
|
+
# The regex engine does not use this. It exists for the judge and for any
|
|
11
|
+
# other tool that asks questions about a paragraph or a sentence rather than
|
|
12
|
+
# matching a pattern across the whole text. See docs/JUDGE.md "Splitting".
|
|
13
|
+
module Split
|
|
14
|
+
# text is the span as the file has it, which is what a note quotes.
|
|
15
|
+
# asked is the same span with what --markdown skips taken out, which is
|
|
16
|
+
# what the model is shown. They are the same string without --markdown.
|
|
17
|
+
Sentence = Data.define(:offset, :length, :text, :asked)
|
|
18
|
+
Paragraph = Data.define(:offset, :length, :text, :asked, :sentences)
|
|
19
|
+
|
|
20
|
+
# Furniture: a line that is not prose, whatever a regex would think.
|
|
21
|
+
# Headings, list items, table rows, block quotes, horizontal rules,
|
|
22
|
+
# reference-style link definitions and a lone image or link line.
|
|
23
|
+
FURNITURE = /\A[ \t]*(?:\#{1,6}[ \t]|[-*+][ \t]|\||>|[-*_]{3,}[ \t]*\z|\[[^\]]+\]:[ \t]|!\[)/
|
|
24
|
+
|
|
25
|
+
# An ordered-list item, which is furniture too but only sometimes: see
|
|
26
|
+
# ordered_item?.
|
|
27
|
+
ORDERED = /\A[ \t]*(\d+)[.)][ \t]/
|
|
28
|
+
|
|
29
|
+
# Abbreviations whose trailing period does not end a sentence. Case
|
|
30
|
+
# matters: "No." is an abbreviation, "no." is the end of a sentence.
|
|
31
|
+
ABBREV = /\b(?:e\.g|i\.e|vs|etc|cf|Fig|No|Dr|Mr|Mrs|Ms|St|Inc|Ltd|Sec|Ch|Vol)\.\z/
|
|
32
|
+
|
|
33
|
+
# The other period that does not end a sentence, matched on its shape
|
|
34
|
+
# rather than on a list of words: letters written one at a time with a
|
|
35
|
+
# period after each, which is "U.S.", "U.K.", "e.g.", "i.e." and the rest
|
|
36
|
+
# of them. Two pairs at least, so the last letter of a word is not one:
|
|
37
|
+
# "bigger than AT&T." and "a free upgrade from 512K." still end their
|
|
38
|
+
# sentences. The run is bounded, so a piece is read in one pass.
|
|
39
|
+
#
|
|
40
|
+
# One period of this shape is left alone: a single capital standing as a
|
|
41
|
+
# word. It is an initial in "J. K. Rowling wrote the book.", which is
|
|
42
|
+
# split into three sentences here, and it is a whole sentence in "Service
|
|
43
|
+
# A. Service B handles the rest.", which a design document writes far more
|
|
44
|
+
# often than it names anyone by initial.
|
|
45
|
+
#
|
|
46
|
+
# What the run costs instead: a sentence that really ends in one of these
|
|
47
|
+
# joins the one after it, so "It was in the U.S. It cost $2." is read as
|
|
48
|
+
# one sentence.
|
|
49
|
+
INITIALS = /(?<![[:alpha:]])(?:[[:alpha:]]\.){2,6}\z/
|
|
50
|
+
|
|
51
|
+
# A number and a period with nothing else between two boundaries is an
|
|
52
|
+
# enumerator, not a sentence: the "2." of a list item that CommonMark
|
|
53
|
+
# keeps inside the paragraph above it, because a list may interrupt a
|
|
54
|
+
# paragraph only when it starts at 1. The whole piece has to be the
|
|
55
|
+
# number, so a year still closes the sentence it sits in -- "It was
|
|
56
|
+
# rewritten in\n2021. Then it shipped." is two sentences.
|
|
57
|
+
ENUMERATOR = /\A\d+\.\z/
|
|
58
|
+
|
|
59
|
+
# The judge's view of Markdown differs from the regex engine's in one
|
|
60
|
+
# way. Fenced code and comments become spaces, as there, but an inline
|
|
61
|
+
# code span or a URL becomes a run of this character, same length: it is
|
|
62
|
+
# a word in its sentence, so "`x` runs fast." starts at the backtick and
|
|
63
|
+
# "Done. `x` runs." is two sentences. Blanked to spaces, the first lost
|
|
64
|
+
# its head and the second merged. It is a private-use codepoint, which
|
|
65
|
+
# ordinary prose cannot contain, so a line of real text is never taken
|
|
66
|
+
# for a blanked one -- a line reading "XXX" used to be.
|
|
67
|
+
INLINE = "\uE000"
|
|
68
|
+
|
|
69
|
+
# A sentence ends at .!? (with an optional closing quote or bracket)
|
|
70
|
+
# followed by whitespace and a capital, digit, opening quote/bracket, or
|
|
71
|
+
# a code span, which starts a sentence as any other word does.
|
|
72
|
+
BOUNDARY = /(?<=[.!?]|[.!?]["'”’)\]])\s+(?=["'“‘(\[A-Z0-9#{INLINE}])/
|
|
73
|
+
|
|
74
|
+
# The split keeps its separators, so one of the blocks it hands back is
|
|
75
|
+
# the break itself. Both of these are written out here rather than inside
|
|
76
|
+
# the walk, where the interpolation would build the same pattern again for
|
|
77
|
+
# every block of every document.
|
|
78
|
+
BREAK_ONLY = /\A#{PARA_BREAK}\z/
|
|
79
|
+
KEEP_BREAK = /(#{PARA_BREAK})/
|
|
80
|
+
KEEP_BOUNDARY = /(#{BOUNDARY})/
|
|
81
|
+
|
|
82
|
+
module_function
|
|
83
|
+
|
|
84
|
+
# text: the source. markdown: blank code, HTML comments and URLs before
|
|
85
|
+
# splitting, and blank furniture lines so they neither count as prose nor
|
|
86
|
+
# take the prose around them with them. Splitting happens on the blanked
|
|
87
|
+
# copy; the texts returned are cut from the original at the same offsets,
|
|
88
|
+
# so an excerpt is always a string that is in the file. Blanking is
|
|
89
|
+
# character for character, so the offsets agree.
|
|
90
|
+
def paragraphs(text, markdown: false)
|
|
91
|
+
scan, asked = markdown ? blank_furniture(blank(text), blank_blocks(text)) : [text, text]
|
|
92
|
+
# A document with nothing to leave out is asked about as written, and
|
|
93
|
+
# then every text is cut once instead of twice.
|
|
94
|
+
asked = text if asked == text
|
|
95
|
+
written = Cursor.new(text)
|
|
96
|
+
shown = asked.equal?(text) ? nil : Cursor.new(asked)
|
|
97
|
+
out = []
|
|
98
|
+
at = 0
|
|
99
|
+
# Split keeping the separators, and walk the offsets arithmetically:
|
|
100
|
+
# String#index with a start position rescans from 0 on non-ASCII text.
|
|
101
|
+
scan.split(KEEP_BREAK).each do |block|
|
|
102
|
+
start = at
|
|
103
|
+
at += block.length
|
|
104
|
+
next if block.match?(BREAK_ONLY)
|
|
105
|
+
|
|
106
|
+
# Both ends trimmed by the same measure. String#strip takes a leading
|
|
107
|
+
# NUL off as well as whitespace and /\A\s*/ does not, so a block that
|
|
108
|
+
# starts with one used to put every offset in the paragraph one
|
|
109
|
+
# character early and cut its last character off.
|
|
110
|
+
from_lead = block.lstrip
|
|
111
|
+
lead = block.length - from_lead.length
|
|
112
|
+
body = from_lead.rstrip
|
|
113
|
+
next if body.empty?
|
|
114
|
+
|
|
115
|
+
offset = start + lead
|
|
116
|
+
# The paragraph is cut out of the original once, and its sentences are
|
|
117
|
+
# cut out of that: cutting each of them out of the whole document by
|
|
118
|
+
# character offset is what made the walk take the square of the
|
|
119
|
+
# document's size.
|
|
120
|
+
para = written.cut(offset, body.length)
|
|
121
|
+
para_asked = shown ? shown.cut(offset, body.length) : para
|
|
122
|
+
sents = sentences(body, base: offset, text: para, asked: para_asked)
|
|
123
|
+
# A one-sentence block ending in a colon is the lead-in to whatever
|
|
124
|
+
# follows (a code block, a list), not a paragraph.
|
|
125
|
+
next if markdown && sents.size == 1 && body.end_with?(":")
|
|
126
|
+
|
|
127
|
+
unless sents.empty?
|
|
128
|
+
flat = squash(para)
|
|
129
|
+
out << Paragraph.new(offset:, length: body.length, text: flat, sentences: sents,
|
|
130
|
+
asked: para.equal?(para_asked) ? flat : squash(para_asked))
|
|
131
|
+
end
|
|
132
|
+
end
|
|
133
|
+
out
|
|
134
|
+
end
|
|
135
|
+
|
|
136
|
+
# Sentences of one paragraph. base is where the paragraph starts in the
|
|
137
|
+
# document, which each sentence's offset is counted from. text is the
|
|
138
|
+
# paragraph as the file has it and asked is the copy the model is shown,
|
|
139
|
+
# both cut at the paragraph's own offsets; a sentence is cut out of them
|
|
140
|
+
# and its whitespace collapsed, so hard-wrapped lines read as one. A
|
|
141
|
+
# caller that hands over a bare paragraph and neither copy gets the body
|
|
142
|
+
# itself as both.
|
|
143
|
+
def sentences(body, base: 0, text: nil, asked: nil)
|
|
144
|
+
text ||= body
|
|
145
|
+
asked ||= text
|
|
146
|
+
# The same cursors as a paragraph is cut with, for the same reason: a
|
|
147
|
+
# document whose sentences all sit in one paragraph is one long string
|
|
148
|
+
# to cut them out of.
|
|
149
|
+
written = Cursor.new(text)
|
|
150
|
+
shown = asked.equal?(text) ? nil : Cursor.new(asked)
|
|
151
|
+
cut = lambda do |at, length|
|
|
152
|
+
flat = squash(written.cut(at, length))
|
|
153
|
+
Sentence.new(offset: base + at, length:, text: flat,
|
|
154
|
+
asked: shown ? squash(shown.cut(at, length)) : flat)
|
|
155
|
+
end
|
|
156
|
+
parts = []
|
|
157
|
+
buf_start = nil
|
|
158
|
+
pos = 0
|
|
159
|
+
body.split(KEEP_BOUNDARY).each do |piece|
|
|
160
|
+
at = pos
|
|
161
|
+
pos = at + piece.length
|
|
162
|
+
next if piece.match?(/\A\s+\z/)
|
|
163
|
+
|
|
164
|
+
buf_start ||= at
|
|
165
|
+
# The abbreviation ends where the piece does, and a piece ends at a
|
|
166
|
+
# .!? that a boundary follows, so it is always inside this piece: the
|
|
167
|
+
# whole buffer never has to be cut out of the body to see it.
|
|
168
|
+
next if piece.match?(ABBREV) || piece.match?(INITIALS) || piece.match?(ENUMERATOR)
|
|
169
|
+
|
|
170
|
+
parts << cut.call(buf_start, pos - buf_start)
|
|
171
|
+
buf_start = nil
|
|
172
|
+
end
|
|
173
|
+
parts << cut.call(buf_start, pos - buf_start) if buf_start
|
|
174
|
+
parts
|
|
175
|
+
end
|
|
176
|
+
|
|
177
|
+
def squash(s) = s.gsub(/\s+/, " ").strip
|
|
178
|
+
|
|
179
|
+
def blank(text)
|
|
180
|
+
text.gsub(Engine::MARKDOWN_NOISE) do |s|
|
|
181
|
+
Regexp.last_match[:inline] ? INLINE * s.length : s.gsub(/[^\n]/, " ")
|
|
182
|
+
end
|
|
183
|
+
end
|
|
184
|
+
|
|
185
|
+
# The copy the model is shown: fenced code and HTML comments go, because
|
|
186
|
+
# --markdown says they do, and a code span or URL stays as written,
|
|
187
|
+
# because it is a word of its sentence and the model has to read it.
|
|
188
|
+
def blank_blocks(text)
|
|
189
|
+
text.gsub(Engine::MARKDOWN_NOISE) do |s|
|
|
190
|
+
Regexp.last_match[:inline] ? s : s.gsub(/[^\n]/, " ")
|
|
191
|
+
end
|
|
192
|
+
end
|
|
193
|
+
|
|
194
|
+
# Blank furniture lines to same-length spaces, and the indented lines that
|
|
195
|
+
# continue a blanked one: a wrapped bullet is one bullet, and its second
|
|
196
|
+
# line is not a sentence of its own. A line that is only code spans or
|
|
197
|
+
# URLs (a command on a line of its own, a bare link) is furniture too.
|
|
198
|
+
LONE_INLINE = /\A[ \t]*#{INLINE}[ \t#{INLINE}]*\z/
|
|
199
|
+
# The same line with sentence punctuation after it. A URL stops before
|
|
200
|
+
# that punctuation now, so the full stop on a bare link line is written
|
|
201
|
+
# there rather than part of the link.
|
|
202
|
+
LONE_INLINE_ENDED = /\A[ \t]*#{INLINE}[ \t#{INLINE}]*[.,;:!?)\]]+\z/
|
|
203
|
+
|
|
204
|
+
# A bullet. FURNITURE matches one too, among everything else it matches;
|
|
205
|
+
# this is here because a bullet and an ordered item are the only two
|
|
206
|
+
# kinds of furniture a continuation line can belong to.
|
|
207
|
+
BULLET = /\A[ \t]*[-*+][ \t]/
|
|
208
|
+
# The second line of a list item, indented under the first.
|
|
209
|
+
CONTINUED = /\A(?:[ ]{2,}|\t)\S/
|
|
210
|
+
# An indented code block: four spaces or a tab. Two spaces are still
|
|
211
|
+
# prose, so an indented paragraph under a heading is kept, which is what
|
|
212
|
+
# CommonMark says as well -- a code block starts at four.
|
|
213
|
+
CODE_INDENT = /\A(?:[ ]{4}|\t)/
|
|
214
|
+
# A line that is one HTML tag, opening or closing, matched on that shape
|
|
215
|
+
# rather than on a list of element names: <details>, <div>, <br/> and
|
|
216
|
+
# whatever else a document drops into Markdown all look the same from
|
|
217
|
+
# here. A sentence that names a tag in running text does not look like
|
|
218
|
+
# this: "<p> is the tag for a paragraph." has words after the ">".
|
|
219
|
+
HTML_LINE = %r{\A[ \t]{0,3}</?[A-Za-z][^\n]*>[ \t]*\z}
|
|
220
|
+
# The --- that opens YAML front matter. FURNITURE reads it as a
|
|
221
|
+
# horizontal rule wherever it appears; front matter is the block it
|
|
222
|
+
# opens, and only on the first line of the file.
|
|
223
|
+
FRONT = /\A---[ \t]*\z/
|
|
224
|
+
|
|
225
|
+
# How many lines of YAML front matter the document opens with: the ---
|
|
226
|
+
# on the first line, everything to the next --- line, and that line. Zero
|
|
227
|
+
# for a document that does not open with one, so a --- between two
|
|
228
|
+
# paragraphs stays the horizontal rule it is.
|
|
229
|
+
def front_matter(lines)
|
|
230
|
+
return 0 unless lines.first&.chomp&.match?(FRONT)
|
|
231
|
+
|
|
232
|
+
close = lines.drop(1).index { |l| l.chomp.match?(FRONT) } or return 0
|
|
233
|
+
|
|
234
|
+
close + 2
|
|
235
|
+
end
|
|
236
|
+
|
|
237
|
+
# Both copies at once, line by line: the furniture is decided on the
|
|
238
|
+
# splitter's copy, where a code span is a placeholder, and the same lines
|
|
239
|
+
# are blanked in the copy the model is shown. Every line keeps its own
|
|
240
|
+
# length and its line ending, so an offset means the same thing in the
|
|
241
|
+
# original and in both copies. The line ending is taken off before the
|
|
242
|
+
# tests run: on a CRLF file it would otherwise sit between the line and
|
|
243
|
+
# the \z that a horizontal rule or a bare link ends at, and neither would
|
|
244
|
+
# be recognised as furniture.
|
|
245
|
+
def blank_furniture(scan, shown)
|
|
246
|
+
lines = scan.each_line.to_a
|
|
247
|
+
front = front_matter(lines)
|
|
248
|
+
item = false
|
|
249
|
+
prose = false
|
|
250
|
+
code = false
|
|
251
|
+
opens = true
|
|
252
|
+
pairs = lines.zip(shown.each_line.to_a).each_with_index.map do |(whole, also), i|
|
|
253
|
+
l = whole.chomp
|
|
254
|
+
ending = whole[l.length..]
|
|
255
|
+
blank = l.strip.empty?
|
|
256
|
+
# Only a list item runs on to the next line. An indented line under a
|
|
257
|
+
# heading, a table row, a horizontal rule or a link definition is an
|
|
258
|
+
# indented paragraph, and chaining from those dropped the prose along
|
|
259
|
+
# with the furniture above it.
|
|
260
|
+
listed = l.match?(BULLET) || ordered_item?(l, prose)
|
|
261
|
+
indented = l.match?(CODE_INDENT)
|
|
262
|
+
# An indented code block opens where a paragraph cannot be running
|
|
263
|
+
# already -- the first line, or after a blank or furniture line --
|
|
264
|
+
# and where the indent is not a list item's second line. It then runs
|
|
265
|
+
# for as long as the indent holds.
|
|
266
|
+
code = (code && (indented || blank)) || (indented && !item && opens)
|
|
267
|
+
dropped = i < front || code || listed || l.match?(FURNITURE) || l.match?(HTML_LINE) ||
|
|
268
|
+
lone_inline?(l, prose) || (item && l.match?(CONTINUED))
|
|
269
|
+
item = listed || (item && l.match?(CONTINUED))
|
|
270
|
+
prose = !dropped && !blank
|
|
271
|
+
opens = blank || dropped
|
|
272
|
+
dropped ? ["#{" " * l.length}#{ending}"] * 2 : [whole, also]
|
|
273
|
+
end
|
|
274
|
+
[pairs.map(&:first).join, pairs.map(&:last).join]
|
|
275
|
+
end
|
|
276
|
+
|
|
277
|
+
# A line that is only code spans or URLs is furniture: a command on a
|
|
278
|
+
# line of its own, a bare link. With sentence punctuation after it, only
|
|
279
|
+
# where no sentence can have been running already -- after a line of
|
|
280
|
+
# prose it is the wrapped tail of that sentence, and "Run it with\n`make
|
|
281
|
+
# test`." is one sentence that would otherwise lose its object.
|
|
282
|
+
def lone_inline?(line, after_prose)
|
|
283
|
+
return true if line.match?(LONE_INLINE)
|
|
284
|
+
|
|
285
|
+
!after_prose && line.match?(LONE_INLINE_ENDED)
|
|
286
|
+
end
|
|
287
|
+
|
|
288
|
+
# CommonMark lets an ordered list interrupt a paragraph only when it
|
|
289
|
+
# starts at 1. So a hard-wrapped prose line that begins with a year --
|
|
290
|
+
# "The library was released in\n2019. It was rewritten in\n2021." -- is
|
|
291
|
+
# prose, not two list items, and the sentences on it stay in view.
|
|
292
|
+
def ordered_item?(line, after_prose)
|
|
293
|
+
m = ORDERED.match(line) or return false
|
|
294
|
+
!after_prose || m[1] == "1"
|
|
295
|
+
end
|
|
296
|
+
|
|
297
|
+
# Cutting a span out of a string by character offset walks the string from
|
|
298
|
+
# its start, because a character is not a fixed number of bytes, so cutting
|
|
299
|
+
# every paragraph of a document out of it one at a time takes the square of
|
|
300
|
+
# the document's size. The spans are asked for in the order they appear, so
|
|
301
|
+
# a cursor holds the byte offset of the last character cut at and answers
|
|
302
|
+
# each one in the length of the span itself: byteslice takes a byte offset,
|
|
303
|
+
# and a byte offset costs nothing to reach.
|
|
304
|
+
class Cursor
|
|
305
|
+
def initialize(text)
|
|
306
|
+
@text = text
|
|
307
|
+
@char = 0
|
|
308
|
+
@byte = 0
|
|
309
|
+
end
|
|
310
|
+
|
|
311
|
+
# The span of length characters at character offset at. A cursor only
|
|
312
|
+
# goes forward: asked to go back it would answer with the wrong span,
|
|
313
|
+
# and a note would quote a string that is not in the file.
|
|
314
|
+
def cut(at, length)
|
|
315
|
+
raise ArgumentError, "a cursor at #{@char} was asked for #{at}" if at < @char
|
|
316
|
+
|
|
317
|
+
advance(at - @char)
|
|
318
|
+
advance(length)
|
|
319
|
+
end
|
|
320
|
+
|
|
321
|
+
private
|
|
322
|
+
|
|
323
|
+
def advance(count)
|
|
324
|
+
return "" unless count.positive?
|
|
325
|
+
|
|
326
|
+
s = window(count)
|
|
327
|
+
@char += s.length
|
|
328
|
+
@byte += s.bytesize
|
|
329
|
+
s
|
|
330
|
+
end
|
|
331
|
+
|
|
332
|
+
# A character is at most four bytes in UTF-8, so four bytes per character
|
|
333
|
+
# is a window wide enough to hold the ones asked for. Any other encoding
|
|
334
|
+
# widens it until it is, or until the string ends. A character split by
|
|
335
|
+
# the far edge of the window is past the count asked for, so it is never
|
|
336
|
+
# one of the ones returned.
|
|
337
|
+
def window(count)
|
|
338
|
+
width = count * 4
|
|
339
|
+
loop do
|
|
340
|
+
s = @text.byteslice(@byte, width)
|
|
341
|
+
return s[0, count] if s.length >= count || s.bytesize < width
|
|
342
|
+
|
|
343
|
+
width *= 2
|
|
344
|
+
end
|
|
345
|
+
end
|
|
346
|
+
end
|
|
347
|
+
end
|
|
348
|
+
end
|
data/lib/sloplint/version.rb
CHANGED
metadata
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
--- !ruby/object:Gem::Specification
|
|
2
2
|
name: sloplint
|
|
3
3
|
version: !ruby/object:Gem::Version
|
|
4
|
-
version: 0.
|
|
4
|
+
version: 0.9.0
|
|
5
5
|
platform: ruby
|
|
6
6
|
authors:
|
|
7
7
|
- Benjamin Jackson
|
|
@@ -58,6 +58,7 @@ files:
|
|
|
58
58
|
- lib/sloplint/engine.rb
|
|
59
59
|
- lib/sloplint/output.rb
|
|
60
60
|
- lib/sloplint/rules.rb
|
|
61
|
+
- lib/sloplint/split.rb
|
|
61
62
|
- lib/sloplint/version.rb
|
|
62
63
|
homepage: https://github.com/benjaminjackson/sloplint
|
|
63
64
|
licenses:
|