leptris 1.9.280.1 → 1.9.282.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
checksums.yaml CHANGED
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  SHA256:
3
- metadata.gz: f8c9a359271bb60e62c4a6fb8e210c725d02b332df53f804411498c82acd1056
4
- data.tar.gz: 5e196178a59975549b08e1c3b54c8d5377d463d430ad8bceef5df410a5ea4e83
3
+ metadata.gz: 2c255f7e95c4ff01c2ff315545555a81472e3f06fae61875db30357b92cb6644
4
+ data.tar.gz: bef13a2b8dab6c2fa220fa244dad937113db907f0cb15940bab2489c8d4c44e8
5
5
  SHA512:
6
- metadata.gz: f4bbbdb0079f57fb234636a4ec03000a44a67ac6f80ea852fc4ba3bac9c0faddce6231900f137c8c9596369f4ea552120be792670da103f5169f2d796c891a56
7
- data.tar.gz: 6a96c92cba79412fd69d32bc33482cd8b208808ded2b7d554539f7322647a9c6852e1531342fd428f7fe5826a94ccbfc428e4d094faca3d251dcdb2bddfb6aec
6
+ metadata.gz: aca607136b2b922b2b330ad743af2e7d847ad1ba0618d60dbdc4401addd0af2d496043d3a1eec2fac984a0934e8c1504a36d8b33b4113a260476f06ef50b442d
7
+ data.tar.gz: 43df939a0c758f17705fc2bff2628f00f0a3bb762e928fc8c61c0a6fabef85920b55a1c60bae98c08e17a674321187f4890bb519857a42b425a0795759b5e629
data/CHANGELOG.md CHANGED
@@ -1,3 +1,27 @@
1
+ ## [1.9.282.0] - 2026-10-01
2
+
3
+ ### Fixed — engine sync (1.9.281)
4
+
5
+ - Vendored libleptris 1.9.281: `fn:upper-case` and `fn:lower-case`
6
+ perform real Unicode case mapping (utf8proc per-codepoint
7
+ `toupper`/`tolower`) instead of ASCII-only `c ± 32` loops —
8
+ `upper-case(U+01CB)` correctly yields U+01CA.
9
+ `leptris_unicode_to_upper`/`to_lower` were case folding in
10
+ disguise and are rewritten as true case mapping;
11
+ `leptris_unicode_casecmp` keeps its genuine fold semantics
12
+ explicitly. Builds without utf8proc keep the ASCII fallback
13
+ (leptris#1182, leptris#1460).
14
+
15
+ ### Added — engine sync (1.9.281-1.9.282)
16
+
17
+ - QT3 string-case corpus family adopted (leptris#1182): fn/upper-case
18
+ (23/30), fn/lower-case (23/28), fn/codepoint-equal (24/37); the
19
+ runner's corpus counts branch on the utf8proc capability.
20
+ - QT3 string-tails batch 2: fn/encode-for-uri adopted — 25 of 30
21
+ cases run-and-agree, pinning the RFC 3986 %XX escaping behavior
22
+ as a permanent regression guard (leptris#1182, leptris#1462).
23
+
24
+
1
25
  ## [1.9.280.1] - 2026-09-30
2
26
 
3
27
  ### Added
data/Rakefile CHANGED
@@ -8,7 +8,7 @@ RSpec::Core::RakeTask.new(:spec)
8
8
  # Pin for `rake compile` and the platform-gem builds. Keep in lockstep
9
9
  # with .github/workflows/build.yml (which calls `rake compile`) and the
10
10
  # CHANGELOG when libleptris releases.
11
- LIBLEPTRIS_VERSION = "1.9.280"
11
+ LIBLEPTRIS_VERSION = "1.9.282"
12
12
  # Vendored alongside libleptris for fn:normalize-unicode (TODO
13
13
  # .restructure/20): built per platform with a RELOCATABLE @rpath
14
14
  # install name, loaded by ffi.rb before libleptris so the
@@ -1,5 +1,5 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module Leptris
4
- VERSION = "1.9.280.1"
4
+ VERSION = "1.9.282.0"
5
5
  end
data/vendor-src/README.md CHANGED
@@ -1,7 +1,7 @@
1
1
  Sources vendored in this gem (recompile rights; the ruby
2
2
  variant builds them at install via extconf.rb):
3
3
 
4
- - libleptris v1.9.280
4
+ - libleptris v1.9.282
5
5
  - utf8proc v2.11.0
6
6
 
7
7
  Rebuild by hand:
@@ -1,5 +1,33 @@
1
1
  ## [Unreleased]
2
2
 
3
+ ## [1.9.282] - 2026-10-01
4
+
5
+ ### Added
6
+
7
+ - QT3 string-tails batch 2: fn/encode-for-uri adopted — 25 of 30 cases run-and-agree on the
8
+ existing implementation (the remainder skip on unsupported environment shapes), pinning the
9
+ RFC 3986 %XX escaping behavior as a permanent regression guard (leptris#1182, leptris#1462).
10
+
11
+ ## [1.9.281] - 2026-09-30
12
+
13
+ ### Fixed
14
+
15
+ - `fn:upper-case` and `fn:lower-case` now perform real Unicode case mapping (utf8proc's
16
+ per-codepoint `toupper`/`tolower`) instead of ASCII-only `c ± 32` loops — `upper-case(U+01CB)`
17
+ correctly yields U+01CA (fn-upper-case-20). `leptris_unicode_to_upper`/`to_lower` were also
18
+ case folding in disguise (`UTF8PROC_CASEFOLD` produces neither upper nor lower) — rewritten as
19
+ true case mapping, with `leptris_unicode_casecmp` keeping its genuine fold semantics
20
+ explicitly. Builds without utf8proc keep the ASCII fallback (leptris#1182, leptris#1460).
21
+
22
+ ### Added
23
+
24
+ - QT3 string-case corpus family adopted (leptris#1182, corpus growth resumed): fn/upper-case
25
+ (23/30), fn/lower-case (23/28), fn/codepoint-equal (24/37). The runner's corpus counts now
26
+ branch on the utf8proc capability — without it the two Unicode-mapping cases are excluded by
27
+ name rather than counted down.
28
+
29
+
30
+
3
31
  ## [1.9.280] - 2026-09-30
4
32
 
5
33
  ### Fixed
@@ -3,7 +3,7 @@
3
3
  cmake_minimum_required(VERSION 3.20)
4
4
 
5
5
  project(leptris
6
- VERSION 1.9.280
6
+ VERSION 1.9.282
7
7
  DESCRIPTION "Fast XML parser and XPath evaluator in pure C"
8
8
  LANGUAGES C CXX
9
9
  )
@@ -108,25 +108,40 @@ char* leptris_unicode_normalize(const char* str, size_t len,
108
108
  /**
109
109
  * Convert UTF-8 string to uppercase
110
110
  */
111
+ /* Unicode CASE MAPPING (F&O upper-case/lower-case): per-codepoint
112
+ * utf8proc_toupper/tolower — NOT case folding. The previous
113
+ * UTF8PROC_CASEFOLD map folded to the canonical (lowercase-shaped)
114
+ * fold form, which is neither operation and left fn:upper-case
115
+ * wrong for every non-ASCII input (fn-upper-case-20: U+01CB must
116
+ * map to U+01CA). Case mapping is codepoint-wise in utf8proc's
117
+ * property API; encode the mapped sequence back to UTF-8. The
118
+ * mapped form of a BMP codepoint is <= 3 bytes, so a 4-byte worst
119
+ * case per input byte is a safe bound (surrogates never appear in
120
+ * valid UTF-8). */
111
121
  char* leptris_unicode_to_upper(const char* str, size_t len, size_t* out_len) {
112
122
  if (!str || !out_len) {
113
123
  return NULL;
114
124
  }
115
125
 
116
- utf8proc_uint8_t* result = NULL;
117
- utf8proc_ssize_t result_len = utf8proc_map(
118
- (const utf8proc_uint8_t*)str,
119
- (utf8proc_ssize_t)len,
120
- &result,
121
- UTF8PROC_STABLE | UTF8PROC_CASEFOLD | UTF8PROC_COMPOSE
122
- );
123
-
124
- if (result_len < 0) {
125
- return NULL;
126
+ char* out = (char*)malloc(len * 4 + 1);
127
+ if (!out) return NULL;
128
+ size_t w = 0;
129
+ for (size_t i = 0; i < len; ) {
130
+ utf8proc_int32_t cp;
131
+ utf8proc_ssize_t used = utf8proc_iterate(
132
+ (const utf8proc_uint8_t*)str + i,
133
+ (utf8proc_ssize_t)(len - i), &cp);
134
+ if (used < 1) { free(out); return NULL; }
135
+ cp = utf8proc_toupper(cp);
136
+ utf8proc_ssize_t written = utf8proc_encode_char(
137
+ cp, (utf8proc_uint8_t*)out + w);
138
+ if (written < 1) { free(out); return NULL; }
139
+ w += (size_t)written;
140
+ i += (size_t)used;
126
141
  }
127
-
128
- *out_len = (size_t)result_len;
129
- return (char*)result;
142
+ out[w] = 0;
143
+ *out_len = w;
144
+ return out;
130
145
  }
131
146
 
132
147
  /**
@@ -137,20 +152,25 @@ char* leptris_unicode_to_lower(const char* str, size_t len, size_t* out_len) {
137
152
  return NULL;
138
153
  }
139
154
 
140
- utf8proc_uint8_t* result = NULL;
141
- utf8proc_ssize_t result_len = utf8proc_map(
142
- (const utf8proc_uint8_t*)str,
143
- (utf8proc_ssize_t)len,
144
- &result,
145
- UTF8PROC_STABLE | UTF8PROC_CASEFOLD | UTF8PROC_COMPOSE
146
- );
147
-
148
- if (result_len < 0) {
149
- return NULL;
155
+ char* out = (char*)malloc(len * 4 + 1);
156
+ if (!out) return NULL;
157
+ size_t w = 0;
158
+ for (size_t i = 0; i < len; ) {
159
+ utf8proc_int32_t cp;
160
+ utf8proc_ssize_t used = utf8proc_iterate(
161
+ (const utf8proc_uint8_t*)str + i,
162
+ (utf8proc_ssize_t)(len - i), &cp);
163
+ if (used < 1) { free(out); return NULL; }
164
+ cp = utf8proc_tolower(cp);
165
+ utf8proc_ssize_t written = utf8proc_encode_char(
166
+ cp, (utf8proc_uint8_t*)out + w);
167
+ if (written < 1) { free(out); return NULL; }
168
+ w += (size_t)written;
169
+ i += (size_t)used;
150
170
  }
151
-
152
- *out_len = (size_t)result_len;
153
- return (char*)result;
171
+ out[w] = 0;
172
+ *out_len = w;
173
+ return out;
154
174
  }
155
175
 
156
176
  /**
@@ -162,10 +182,25 @@ int leptris_unicode_casecmp(const char* str1, size_t len1,
162
182
  return (str1 == str2) ? 0 : (str1 ? 1 : -1);
163
183
  }
164
184
 
165
- /* Normalize both strings to lowercase for comparison */
185
+ /* Fold both strings for comparison — case folding, NOT
186
+ * lower-case mapping: the fold form is the canonical
187
+ * case-insensitive identity (the old to_lower did this
188
+ * implicitly; true case mapping is not a substitute). */
166
189
  size_t norm1_len, norm2_len;
167
- char* norm1 = leptris_unicode_to_lower(str1, len1, &norm1_len);
168
- char* norm2 = leptris_unicode_to_lower(str2, len2, &norm2_len);
190
+ utf8proc_uint8_t *f1 = NULL, *f2 = NULL;
191
+ utf8proc_ssize_t n1 = utf8proc_map(
192
+ (const utf8proc_uint8_t*)str1, (utf8proc_ssize_t)len1, &f1,
193
+ UTF8PROC_STABLE | UTF8PROC_CASEFOLD | UTF8PROC_COMPOSE);
194
+ utf8proc_ssize_t n2 = utf8proc_map(
195
+ (const utf8proc_uint8_t*)str2, (utf8proc_ssize_t)len2, &f2,
196
+ UTF8PROC_STABLE | UTF8PROC_CASEFOLD | UTF8PROC_COMPOSE);
197
+ if (n1 < 0 || n2 < 0) {
198
+ free(f1); free(f2);
199
+ return (str1 == str2) ? 0 : 1;
200
+ }
201
+ char* norm1 = (char*)f1;
202
+ char* norm2 = (char*)f2;
203
+ (void)0;
169
204
 
170
205
  if (!norm1 || !norm2) {
171
206
  free(norm1);
@@ -15,6 +15,7 @@
15
15
  #include "../dom/cdata.h"
16
16
  #include "../dom/comment.h"
17
17
  #include "../common/port.h"
18
+ #include "../unicode/unicode.h"
18
19
  #include "../dtd/model.h" /* ttdtd_lookup_attribute (id() §4.1) */
19
20
  #include <string.h>
20
21
  #include <stdlib.h>
@@ -2708,14 +2709,24 @@ static struct leptris_xpath_result* xpath_func_upper_case(
2708
2709
  char* src = result_to_string(arg);
2709
2710
  xpath_result_free(arg);
2710
2711
  if (!src) return NULL;
2712
+ /* Unicode case mapping (F&O §7.4.7); ASCII fallback when
2713
+ * utf8proc is compiled out (the cargo legs build without it —
2714
+ * unicode.c is excluded there, so the helper must not be
2715
+ * referenced unguarded). */
2711
2716
  size_t n = strlen(src);
2712
- char* out = LEPTRIS_ALLOC_N(char, n + 1);
2713
- if (!out) { LEPTRIS_FREE(src); return NULL; }
2714
- for (size_t i = 0; i < n; i++) {
2715
- unsigned char c = (unsigned char)src[i];
2716
- out[i] = (char)((c >= 'a' && c <= 'z') ? c - 32 : c);
2717
+ char* out = NULL;
2718
+ #ifdef LEPTRIS_HAS_UTF8PROC
2719
+ out = leptris_unicode_to_upper(src, n, &n);
2720
+ #endif
2721
+ if (!out) {
2722
+ out = LEPTRIS_ALLOC_N(char, n + 1);
2723
+ if (!out) { LEPTRIS_FREE(src); return NULL; }
2724
+ for (size_t i = 0; i < n; i++) {
2725
+ unsigned char c = (unsigned char)src[i];
2726
+ out[i] = (char)((c >= 'a' && c <= 'z') ? c - 32 : c);
2727
+ }
2728
+ out[n] = 0;
2717
2729
  }
2718
- out[n] = 0;
2719
2730
  LEPTRIS_FREE(src);
2720
2731
  struct leptris_xpath_result* result =
2721
2732
  xpath_result_new(XPATH_RESULT_STRING);
@@ -2734,13 +2745,19 @@ static struct leptris_xpath_result* xpath_func_lower_case(
2734
2745
  xpath_result_free(arg);
2735
2746
  if (!src) return NULL;
2736
2747
  size_t n = strlen(src);
2737
- char* out = LEPTRIS_ALLOC_N(char, n + 1);
2738
- if (!out) { LEPTRIS_FREE(src); return NULL; }
2739
- for (size_t i = 0; i < n; i++) {
2740
- unsigned char c = (unsigned char)src[i];
2741
- out[i] = (char)((c >= 'A' && c <= 'Z') ? c + 32 : c);
2748
+ char* out = NULL;
2749
+ #ifdef LEPTRIS_HAS_UTF8PROC
2750
+ out = leptris_unicode_to_lower(src, n, &n);
2751
+ #endif
2752
+ if (!out) {
2753
+ out = LEPTRIS_ALLOC_N(char, n + 1);
2754
+ if (!out) { LEPTRIS_FREE(src); return NULL; }
2755
+ for (size_t i = 0; i < n; i++) {
2756
+ unsigned char c = (unsigned char)src[i];
2757
+ out[i] = (char)((c >= 'A' && c <= 'Z') ? c + 32 : c);
2758
+ }
2759
+ out[n] = 0;
2742
2760
  }
2743
- out[n] = 0;
2744
2761
  LEPTRIS_FREE(src);
2745
2762
  struct leptris_xpath_result* result =
2746
2763
  xpath_result_new(XPATH_RESULT_STRING);
@@ -87,6 +87,13 @@ leptris_add_test(test_sch_corpus sch/test_schematron_corpus.cpp)
87
87
  leptris_add_test(test_qt3 xquery/test_qt3.cpp)
88
88
  target_compile_definitions(test_qt3 PRIVATE
89
89
  LEPTRIS_QT3_DIR="${CMAKE_CURRENT_SOURCE_DIR}/xquery/qt3")
90
+ # leptris_static aggregates leptris_objects as SOURCES, so the
91
+ # objects' INTERFACE compile definition does not flow through to
92
+ # test consumers — mirror the capability flag the runner's corpus
93
+ # counts branch on.
94
+ if(LEPTRIS_ENABLE_UTF8PROC)
95
+ target_compile_definitions(test_qt3 PRIVATE LEPTRIS_HAS_UTF8PROC=1)
96
+ endif()
90
97
  target_compile_definitions(test_sch_corpus PRIVATE
91
98
  LEPTRIS_SCH_CASES_DIR="${CMAKE_CURRENT_SOURCE_DIR}/sch/conformance-cases")
92
99
  leptris_add_test(test_parser parser/test_parser.cpp)