sloplint 0.8.0 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,348 @@
1
+ # frozen_string_literal: true
2
+
3
+ require_relative "rules"
4
+ require_relative "engine"
5
+
6
+ module Sloplint
7
+ # Splits a document into paragraphs and sentences, keeping each one's offset
8
+ # and length in the original text so a note can land on the file as written.
9
+ #
10
+ # The regex engine does not use this. It exists for the judge and for any
11
+ # other tool that asks questions about a paragraph or a sentence rather than
12
+ # matching a pattern across the whole text. See docs/JUDGE.md "Splitting".
13
+ module Split
14
+ # text is the span as the file has it, which is what a note quotes.
15
+ # asked is the same span with what --markdown skips taken out, which is
16
+ # what the model is shown. They are the same string without --markdown.
17
+ Sentence = Data.define(:offset, :length, :text, :asked)
18
+ Paragraph = Data.define(:offset, :length, :text, :asked, :sentences)
19
+
20
+ # Furniture: a line that is not prose, whatever a regex would think.
21
+ # Headings, list items, table rows, block quotes, horizontal rules,
22
+ # reference-style link definitions and a lone image or link line.
23
+ FURNITURE = /\A[ \t]*(?:\#{1,6}[ \t]|[-*+][ \t]|\||>|[-*_]{3,}[ \t]*\z|\[[^\]]+\]:[ \t]|!\[)/
24
+
25
+ # An ordered-list item, which is furniture too but only sometimes: see
26
+ # ordered_item?.
27
+ ORDERED = /\A[ \t]*(\d+)[.)][ \t]/
28
+
29
+ # Abbreviations whose trailing period does not end a sentence. Case
30
+ # matters: "No." is an abbreviation, "no." is the end of a sentence.
31
+ ABBREV = /\b(?:e\.g|i\.e|vs|etc|cf|Fig|No|Dr|Mr|Mrs|Ms|St|Inc|Ltd|Sec|Ch|Vol)\.\z/
32
+
33
+ # The other period that does not end a sentence, matched on its shape
34
+ # rather than on a list of words: letters written one at a time with a
35
+ # period after each, which is "U.S.", "U.K.", "e.g.", "i.e." and the rest
36
+ # of them. Two pairs at least, so the last letter of a word is not one:
37
+ # "bigger than AT&T." and "a free upgrade from 512K." still end their
38
+ # sentences. The run is bounded, so a piece is read in one pass.
39
+ #
40
+ # One period of this shape is left alone: a single capital standing as a
41
+ # word. It is an initial in "J. K. Rowling wrote the book.", which is
42
+ # split into three sentences here, and it is a whole sentence in "Service
43
+ # A. Service B handles the rest.", which a design document writes far more
44
+ # often than it names anyone by initial.
45
+ #
46
+ # What the run costs instead: a sentence that really ends in one of these
47
+ # joins the one after it, so "It was in the U.S. It cost $2." is read as
48
+ # one sentence.
49
+ INITIALS = /(?<![[:alpha:]])(?:[[:alpha:]]\.){2,6}\z/
50
+
51
+ # A number and a period with nothing else between two boundaries is an
52
+ # enumerator, not a sentence: the "2." of a list item that CommonMark
53
+ # keeps inside the paragraph above it, because a list may interrupt a
54
+ # paragraph only when it starts at 1. The whole piece has to be the
55
+ # number, so a year still closes the sentence it sits in -- "It was
56
+ # rewritten in\n2021. Then it shipped." is two sentences.
57
+ ENUMERATOR = /\A\d+\.\z/
58
+
59
+ # The judge's view of Markdown differs from the regex engine's in one
60
+ # way. Fenced code and comments become spaces, as there, but an inline
61
+ # code span or a URL becomes a run of this character, same length: it is
62
+ # a word in its sentence, so "`x` runs fast." starts at the backtick and
63
+ # "Done. `x` runs." is two sentences. Blanked to spaces, the first lost
64
+ # its head and the second merged. It is a private-use codepoint, which
65
+ # ordinary prose cannot contain, so a line of real text is never taken
66
+ # for a blanked one -- a line reading "XXX" used to be.
67
+ INLINE = "\uE000"
68
+
69
+ # A sentence ends at .!? (with an optional closing quote or bracket)
70
+ # followed by whitespace and a capital, digit, opening quote/bracket, or
71
+ # a code span, which starts a sentence as any other word does.
72
+ BOUNDARY = /(?<=[.!?]|[.!?]["'”’)\]])\s+(?=["'“‘(\[A-Z0-9#{INLINE}])/
73
+
74
+ # The split keeps its separators, so one of the blocks it hands back is
75
+ # the break itself. Both of these are written out here rather than inside
76
+ # the walk, where the interpolation would build the same pattern again for
77
+ # every block of every document.
78
+ BREAK_ONLY = /\A#{PARA_BREAK}\z/
79
+ KEEP_BREAK = /(#{PARA_BREAK})/
80
+ KEEP_BOUNDARY = /(#{BOUNDARY})/
81
+
82
+ module_function
83
+
84
+ # text: the source. markdown: blank code, HTML comments and URLs before
85
+ # splitting, and blank furniture lines so they neither count as prose nor
86
+ # take the prose around them with them. Splitting happens on the blanked
87
+ # copy; the texts returned are cut from the original at the same offsets,
88
+ # so an excerpt is always a string that is in the file. Blanking is
89
+ # character for character, so the offsets agree.
90
+ def paragraphs(text, markdown: false)
91
+ scan, asked = markdown ? blank_furniture(blank(text), blank_blocks(text)) : [text, text]
92
+ # A document with nothing to leave out is asked about as written, and
93
+ # then every text is cut once instead of twice.
94
+ asked = text if asked == text
95
+ written = Cursor.new(text)
96
+ shown = asked.equal?(text) ? nil : Cursor.new(asked)
97
+ out = []
98
+ at = 0
99
+ # Split keeping the separators, and walk the offsets arithmetically:
100
+ # String#index with a start position rescans from 0 on non-ASCII text.
101
+ scan.split(KEEP_BREAK).each do |block|
102
+ start = at
103
+ at += block.length
104
+ next if block.match?(BREAK_ONLY)
105
+
106
+ # Both ends trimmed by the same measure. String#strip takes a leading
107
+ # NUL off as well as whitespace and /\A\s*/ does not, so a block that
108
+ # starts with one used to put every offset in the paragraph one
109
+ # character early and cut its last character off.
110
+ from_lead = block.lstrip
111
+ lead = block.length - from_lead.length
112
+ body = from_lead.rstrip
113
+ next if body.empty?
114
+
115
+ offset = start + lead
116
+ # The paragraph is cut out of the original once, and its sentences are
117
+ # cut out of that: cutting each of them out of the whole document by
118
+ # character offset is what made the walk take the square of the
119
+ # document's size.
120
+ para = written.cut(offset, body.length)
121
+ para_asked = shown ? shown.cut(offset, body.length) : para
122
+ sents = sentences(body, base: offset, text: para, asked: para_asked)
123
+ # A one-sentence block ending in a colon is the lead-in to whatever
124
+ # follows (a code block, a list), not a paragraph.
125
+ next if markdown && sents.size == 1 && body.end_with?(":")
126
+
127
+ unless sents.empty?
128
+ flat = squash(para)
129
+ out << Paragraph.new(offset:, length: body.length, text: flat, sentences: sents,
130
+ asked: para.equal?(para_asked) ? flat : squash(para_asked))
131
+ end
132
+ end
133
+ out
134
+ end
135
+
136
+ # Sentences of one paragraph. base is where the paragraph starts in the
137
+ # document, which each sentence's offset is counted from. text is the
138
+ # paragraph as the file has it and asked is the copy the model is shown,
139
+ # both cut at the paragraph's own offsets; a sentence is cut out of them
140
+ # and its whitespace collapsed, so hard-wrapped lines read as one. A
141
+ # caller that hands over a bare paragraph and neither copy gets the body
142
+ # itself as both.
143
+ def sentences(body, base: 0, text: nil, asked: nil)
144
+ text ||= body
145
+ asked ||= text
146
+ # The same cursors as a paragraph is cut with, for the same reason: a
147
+ # document whose sentences all sit in one paragraph is one long string
148
+ # to cut them out of.
149
+ written = Cursor.new(text)
150
+ shown = asked.equal?(text) ? nil : Cursor.new(asked)
151
+ cut = lambda do |at, length|
152
+ flat = squash(written.cut(at, length))
153
+ Sentence.new(offset: base + at, length:, text: flat,
154
+ asked: shown ? squash(shown.cut(at, length)) : flat)
155
+ end
156
+ parts = []
157
+ buf_start = nil
158
+ pos = 0
159
+ body.split(KEEP_BOUNDARY).each do |piece|
160
+ at = pos
161
+ pos = at + piece.length
162
+ next if piece.match?(/\A\s+\z/)
163
+
164
+ buf_start ||= at
165
+ # The abbreviation ends where the piece does, and a piece ends at a
166
+ # .!? that a boundary follows, so it is always inside this piece: the
167
+ # whole buffer never has to be cut out of the body to see it.
168
+ next if piece.match?(ABBREV) || piece.match?(INITIALS) || piece.match?(ENUMERATOR)
169
+
170
+ parts << cut.call(buf_start, pos - buf_start)
171
+ buf_start = nil
172
+ end
173
+ parts << cut.call(buf_start, pos - buf_start) if buf_start
174
+ parts
175
+ end
176
+
177
+ def squash(s) = s.gsub(/\s+/, " ").strip
178
+
179
+ def blank(text)
180
+ text.gsub(Engine::MARKDOWN_NOISE) do |s|
181
+ Regexp.last_match[:inline] ? INLINE * s.length : s.gsub(/[^\n]/, " ")
182
+ end
183
+ end
184
+
185
+ # The copy the model is shown: fenced code and HTML comments go, because
186
+ # --markdown says they do, and a code span or URL stays as written,
187
+ # because it is a word of its sentence and the model has to read it.
188
+ def blank_blocks(text)
189
+ text.gsub(Engine::MARKDOWN_NOISE) do |s|
190
+ Regexp.last_match[:inline] ? s : s.gsub(/[^\n]/, " ")
191
+ end
192
+ end
193
+
194
+ # Blank furniture lines to same-length spaces, and the indented lines that
195
+ # continue a blanked one: a wrapped bullet is one bullet, and its second
196
+ # line is not a sentence of its own. A line that is only code spans or
197
+ # URLs (a command on a line of its own, a bare link) is furniture too.
198
+ LONE_INLINE = /\A[ \t]*#{INLINE}[ \t#{INLINE}]*\z/
199
+ # The same line with sentence punctuation after it. A URL stops before
200
+ # that punctuation now, so the full stop on a bare link line is written
201
+ # there rather than part of the link.
202
+ LONE_INLINE_ENDED = /\A[ \t]*#{INLINE}[ \t#{INLINE}]*[.,;:!?)\]]+\z/
203
+
204
+ # A bullet. FURNITURE matches one too, among everything else it matches;
205
+ # this is here because a bullet and an ordered item are the only two
206
+ # kinds of furniture a continuation line can belong to.
207
+ BULLET = /\A[ \t]*[-*+][ \t]/
208
+ # The second line of a list item, indented under the first.
209
+ CONTINUED = /\A(?:[ ]{2,}|\t)\S/
210
+ # An indented code block: four spaces or a tab. Two spaces are still
211
+ # prose, so an indented paragraph under a heading is kept, which is what
212
+ # CommonMark says as well -- a code block starts at four.
213
+ CODE_INDENT = /\A(?:[ ]{4}|\t)/
214
+ # A line that is one HTML tag, opening or closing, matched on that shape
215
+ # rather than on a list of element names: <details>, <div>, <br/> and
216
+ # whatever else a document drops into Markdown all look the same from
217
+ # here. A sentence that names a tag in running text does not look like
218
+ # this: "<p> is the tag for a paragraph." has words after the ">".
219
+ HTML_LINE = %r{\A[ \t]{0,3}</?[A-Za-z][^\n]*>[ \t]*\z}
220
+ # The --- that opens YAML front matter. FURNITURE reads it as a
221
+ # horizontal rule wherever it appears; front matter is the block it
222
+ # opens, and only on the first line of the file.
223
+ FRONT = /\A---[ \t]*\z/
224
+
225
+ # How many lines of YAML front matter the document opens with: the ---
226
+ # on the first line, everything to the next --- line, and that line. Zero
227
+ # for a document that does not open with one, so a --- between two
228
+ # paragraphs stays the horizontal rule it is.
229
+ def front_matter(lines)
230
+ return 0 unless lines.first&.chomp&.match?(FRONT)
231
+
232
+ close = lines.drop(1).index { |l| l.chomp.match?(FRONT) } or return 0
233
+
234
+ close + 2
235
+ end
236
+
237
+ # Both copies at once, line by line: the furniture is decided on the
238
+ # splitter's copy, where a code span is a placeholder, and the same lines
239
+ # are blanked in the copy the model is shown. Every line keeps its own
240
+ # length and its line ending, so an offset means the same thing in the
241
+ # original and in both copies. The line ending is taken off before the
242
+ # tests run: on a CRLF file it would otherwise sit between the line and
243
+ # the \z that a horizontal rule or a bare link ends at, and neither would
244
+ # be recognised as furniture.
245
+ def blank_furniture(scan, shown)
246
+ lines = scan.each_line.to_a
247
+ front = front_matter(lines)
248
+ item = false
249
+ prose = false
250
+ code = false
251
+ opens = true
252
+ pairs = lines.zip(shown.each_line.to_a).each_with_index.map do |(whole, also), i|
253
+ l = whole.chomp
254
+ ending = whole[l.length..]
255
+ blank = l.strip.empty?
256
+ # Only a list item runs on to the next line. An indented line under a
257
+ # heading, a table row, a horizontal rule or a link definition is an
258
+ # indented paragraph, and chaining from those dropped the prose along
259
+ # with the furniture above it.
260
+ listed = l.match?(BULLET) || ordered_item?(l, prose)
261
+ indented = l.match?(CODE_INDENT)
262
+ # An indented code block opens where a paragraph cannot be running
263
+ # already -- the first line, or after a blank or furniture line --
264
+ # and where the indent is not a list item's second line. It then runs
265
+ # for as long as the indent holds.
266
+ code = (code && (indented || blank)) || (indented && !item && opens)
267
+ dropped = i < front || code || listed || l.match?(FURNITURE) || l.match?(HTML_LINE) ||
268
+ lone_inline?(l, prose) || (item && l.match?(CONTINUED))
269
+ item = listed || (item && l.match?(CONTINUED))
270
+ prose = !dropped && !blank
271
+ opens = blank || dropped
272
+ dropped ? ["#{" " * l.length}#{ending}"] * 2 : [whole, also]
273
+ end
274
+ [pairs.map(&:first).join, pairs.map(&:last).join]
275
+ end
276
+
277
+ # A line that is only code spans or URLs is furniture: a command on a
278
+ # line of its own, a bare link. With sentence punctuation after it, only
279
+ # where no sentence can have been running already -- after a line of
280
+ # prose it is the wrapped tail of that sentence, and "Run it with\n`make
281
+ # test`." is one sentence that would otherwise lose its object.
282
+ def lone_inline?(line, after_prose)
283
+ return true if line.match?(LONE_INLINE)
284
+
285
+ !after_prose && line.match?(LONE_INLINE_ENDED)
286
+ end
287
+
288
+ # CommonMark lets an ordered list interrupt a paragraph only when it
289
+ # starts at 1. So a hard-wrapped prose line that begins with a year --
290
+ # "The library was released in\n2019. It was rewritten in\n2021." -- is
291
+ # prose, not two list items, and the sentences on it stay in view.
292
+ def ordered_item?(line, after_prose)
293
+ m = ORDERED.match(line) or return false
294
+ !after_prose || m[1] == "1"
295
+ end
296
+
297
+ # Cutting a span out of a string by character offset walks the string from
298
+ # its start, because a character is not a fixed number of bytes, so cutting
299
+ # every paragraph of a document out of it one at a time takes the square of
300
+ # the document's size. The spans are asked for in the order they appear, so
301
+ # a cursor holds the byte offset of the last character cut at and answers
302
+ # each one in the length of the span itself: byteslice takes a byte offset,
303
+ # and a byte offset costs nothing to reach.
304
+ class Cursor
305
+ def initialize(text)
306
+ @text = text
307
+ @char = 0
308
+ @byte = 0
309
+ end
310
+
311
+ # The span of length characters at character offset at. A cursor only
312
+ # goes forward: asked to go back it would answer with the wrong span,
313
+ # and a note would quote a string that is not in the file.
314
+ def cut(at, length)
315
+ raise ArgumentError, "a cursor at #{@char} was asked for #{at}" if at < @char
316
+
317
+ advance(at - @char)
318
+ advance(length)
319
+ end
320
+
321
+ private
322
+
323
+ def advance(count)
324
+ return "" unless count.positive?
325
+
326
+ s = window(count)
327
+ @char += s.length
328
+ @byte += s.bytesize
329
+ s
330
+ end
331
+
332
+ # A character is at most four bytes in UTF-8, so four bytes per character
333
+ # is a window wide enough to hold the ones asked for. Any other encoding
334
+ # widens it until it is, or until the string ends. A character split by
335
+ # the far edge of the window is past the count asked for, so it is never
336
+ # one of the ones returned.
337
+ def window(count)
338
+ width = count * 4
339
+ loop do
340
+ s = @text.byteslice(@byte, width)
341
+ return s[0, count] if s.length >= count || s.bytesize < width
342
+
343
+ width *= 2
344
+ end
345
+ end
346
+ end
347
+ end
348
+ end
@@ -1,5 +1,5 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module Sloplint
4
- VERSION = "0.8.0"
4
+ VERSION = "0.9.0"
5
5
  end
metadata CHANGED
@@ -1,7 +1,7 @@
1
1
  --- !ruby/object:Gem::Specification
2
2
  name: sloplint
3
3
  version: !ruby/object:Gem::Version
4
- version: 0.8.0
4
+ version: 0.9.0
5
5
  platform: ruby
6
6
  authors:
7
7
  - Benjamin Jackson
@@ -58,6 +58,7 @@ files:
58
58
  - lib/sloplint/engine.rb
59
59
  - lib/sloplint/output.rb
60
60
  - lib/sloplint/rules.rb
61
+ - lib/sloplint/split.rb
61
62
  - lib/sloplint/version.rb
62
63
  homepage: https://github.com/benjaminjackson/sloplint
63
64
  licenses: