canon 0.3.40 → 0.3.41

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
checksums.yaml CHANGED
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  SHA256:
3
- metadata.gz: 7ceb43d90065cc39a64f8c74b6317dd9f8752aaa807ec7869ae3188c9e494b43
4
- data.tar.gz: d6375f29977244331faff0d8dd54396d52494133d77e4286011804a63bd1d23f
3
+ metadata.gz: 3b6ae46cc828d11dd9474dbaad7453e48cd8c6282d6523732683e06812c919ba
4
+ data.tar.gz: d9245d48440921bca331af7c3be00843c21647cbd043473580c6a218f6a15f77
5
5
  SHA512:
6
- metadata.gz: bed4791632c2d75f6db4e89e5c991b05a92593fdfaa16d7b20c21dbf5be3070914b0b4214567e9ca3045db8f9d3a8312a218acca28574214d76d817ee09706fd
7
- data.tar.gz: 0d3f8b7bad01bd3bd1e089f5488dac9cd23ca9761ab287ad3f5f0340e918262e5b4fcf63c4bb64ddb427e19c3e1bec878b0bef9aaf93b596c5a2a1f125dadd99
6
+ metadata.gz: 7e9eb37f4285e047e2defa49566b19f3d2ddec40cb1fe55f6711c16674dcd076a805244c1716996a442b1bac99dc44af8e2c56ef89c49d677a5cf6649096adcc
7
+ data.tar.gz: f39f78aae5823e33033d328138da4dc5a748c3ee1c55079190457f926bcebeea4767ffc5e515132cd3e767347d80b4065c7fdd1d8e60e7a0833306053271b2e3
@@ -1019,7 +1019,9 @@ module Canon
1019
1019
  opener = /<#{element_name}[>\s]/
1020
1020
  occurrences = SourceLocator.locate_all(value, text, line_map)
1021
1021
  occurrences.each do |occ|
1022
- count = text[0...occ[:char_offset]].scan(opener).length
1022
+ count = count_elements_before_position_open(text,
1023
+ occ[:char_offset],
1024
+ opener)
1023
1025
  return occ if count == target_index
1024
1026
  end
1025
1027
 
@@ -1445,9 +1447,29 @@ range_start, range_end)
1445
1447
  # @param char_offset [Integer] character offset to check before
1446
1448
  # @param element_name [String] name of element to count
1447
1449
  # @return [Integer] element index (0-based) of the element containing the position
1450
+ # Occurrence count with offsets AT the element's own opening
1451
+ # tag (no inside-the-element correction — see
1452
+ # locate_element_at_index).
1453
+ def count_elements_before_position_open(text, char_offset, opener)
1454
+ count = 0
1455
+ pos = 0
1456
+ while (hit = text.index(opener, pos)) && hit < char_offset
1457
+ count += 1
1458
+ pos = hit + 1
1459
+ end
1460
+ count
1461
+ end
1462
+
1448
1463
  def count_elements_before_position(text, char_offset, element_name)
1449
- prefix = text[0...char_offset]
1450
- count = prefix.scan(/<#{element_name}[>\s]/).length
1464
+ # Index loop instead of copying the prefix per call — the
1465
+ # copy was O(offset) on every occurrence check.
1466
+ opener = /<#{element_name}[>\s]/
1467
+ count = 0
1468
+ pos = 0
1469
+ while (hit = text.index(opener, pos)) && hit < char_offset
1470
+ count += 1
1471
+ pos = hit + 1
1472
+ end
1451
1473
  # Subtract 1 because the count includes the element we are inside
1452
1474
  [count - 1, 0].max
1453
1475
  end
@@ -113,8 +113,7 @@ module Canon
113
113
  def self.needs_escaping?(text)
114
114
  return false if text.nil?
115
115
 
116
- text.each_char.any? do |c|
117
- codepoint = c.ord
116
+ text.each_codepoint.any? do |codepoint|
118
117
  codepoint < 32 || codepoint >= 127 || codepoint == 34 || codepoint == 92
119
118
  end
120
119
  end
@@ -12,12 +12,23 @@ module Canon
12
12
  # @param visualization_map [Hash] Character visualization map
13
13
  # @return [Hash] Hash of characters with their metadata
14
14
  def self.detect_non_ascii(text, visualization_map)
15
+ # each_char allocates a 1-char String per character — over two
16
+ # full documents per render that is the presentation stage's
17
+ # biggest allocation source. ASCII-only text (the common case)
18
+ # needs no scan at all; otherwise scan codepoints and
19
+ # materialize the character String only at non-ASCII offsets.
20
+ return {} if text.ascii_only?
21
+
15
22
  detected = {}
16
23
  category_map = DiffFormatter::CHARACTER_CATEGORY_MAP
17
24
  metadata = DiffFormatter::CHARACTER_METADATA
18
25
 
19
- text.each_char do |char|
20
- next if char.ord <= 127
26
+ index = -1
27
+ text.each_codepoint do |codepoint|
28
+ index += 1
29
+ next if codepoint <= 127
30
+
31
+ char = text[index, 1]
21
32
  next if detected.key?(char)
22
33
 
23
34
  visualization = visualization_map.fetch(char, char)
data/lib/canon/version.rb CHANGED
@@ -1,5 +1,5 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module Canon
4
- VERSION = "0.3.40"
4
+ VERSION = "0.3.41"
5
5
  end
metadata CHANGED
@@ -1,7 +1,7 @@
1
1
  --- !ruby/object:Gem::Specification
2
2
  name: canon
3
3
  version: !ruby/object:Gem::Version
4
- version: 0.3.40
4
+ version: 0.3.41
5
5
  platform: ruby
6
6
  authors:
7
7
  - Ribose Inc.