data_redactor 0.9.0 → 0.10.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -3,22 +3,12 @@
3
3
  #include "placeholder.h"
4
4
  #include "custom_patterns.h"
5
5
  #include "redact.h"
6
+ #include "matcher.h"
7
+ #include "tags.h"
6
8
  #include <regex.h>
7
9
  #include <string.h>
8
10
  #include <stdlib.h>
9
11
 
10
- /*
11
- * To map working-buffer positions back to original-string positions we
12
- * maintain a log of every replacement already applied. Each entry records
13
- * where in the *working* buffer the replacement started (after all prior
14
- * replacements) and how many bytes were removed (orig_len) vs. inserted
15
- * (always 10, the length of "[REDACTED]").
16
- *
17
- * For a new match at working position W:
18
- * cumulative_shift_before_W = sum of (10 - orig_len) for all prior
19
- * replacements whose working_pos <= W
20
- * original_pos = W - cumulative_shift_before_W
21
- */
22
12
  /* Look up the i-th entry of the enable_bits Array. Out-of-bounds → 0 (skip). */
23
13
  static inline int scan_enable_bit(VALUE rb_enable_bits, long i) {
24
14
  if (i < 0 || i >= RARRAY_LEN(rb_enable_bits)) return 0;
@@ -26,105 +16,164 @@ static inline int scan_enable_bit(VALUE rb_enable_bits, long i) {
26
16
  return RTEST(v) && NUM2INT(v) != 0;
27
17
  }
28
18
 
19
+ /*
20
+ * Map a working-buffer position (after built-in redaction) back to the
21
+ * original-input position.
22
+ *
23
+ * After the built-in pass, the working buffer contains the original input
24
+ * with each matched CORE span replaced by "[REDACTED]" (10 bytes). The
25
+ * ev[] array (sorted by start, non-overlapping) records every replacement
26
+ * in original-frame coordinates. Walking ev[] we can find which verbatim
27
+ * segment or replacement a working position falls in and recover the
28
+ * original position.
29
+ *
30
+ * For a match that lands inside a "[REDACTED]" span we return the start of
31
+ * the corresponding original CORE (can only happen if a custom pattern
32
+ * matches the literal "[REDACTED]" itself, which is a degenerate case).
33
+ */
34
+ static long working_to_orig(long wpos, const mm_match_t *ev, size_t n,
35
+ size_t ph_len) {
36
+ long cum_orig = 0;
37
+ long cum_work = 0;
38
+ for (size_t i = 0; i < n; i++) {
39
+ long seg = (long)ev[i].start - cum_orig;
40
+ if (wpos < cum_work + seg)
41
+ return cum_orig + (wpos - cum_work);
42
+ cum_orig += seg + (long)ev[i].length;
43
+ cum_work += seg + (long)ph_len;
44
+ }
45
+ return cum_orig + (wpos - cum_work);
46
+ }
47
+
29
48
  VALUE rb_data_redactor_scan(VALUE self, VALUE rb_text, VALUE rb_enable_bits) {
30
49
  Check_Type(rb_text, T_STRING);
31
50
  Check_Type(rb_enable_bits, T_ARRAY);
32
51
 
33
- const char *input = StringValueCStr(rb_text);
52
+ const char *input = RSTRING_PTR(rb_text);
53
+ size_t in_len = (size_t)RSTRING_LEN(rb_text);
54
+
55
+ static const placeholder_t ph_plain = { PLACEHOLDER_MODE_PLAIN, "[REDACTED]" };
56
+
57
+ /* ------------------------------------------------------------------ */
58
+ /* Stage 1: built-ins through v19 (original-frame coords, no rewrite */
59
+ /* coordinate mapping needed). */
60
+ /* ------------------------------------------------------------------ */
34
61
 
35
- static const placeholder_t ph_default = { PLACEHOLDER_MODE_PLAIN, "[REDACTED]" };
62
+ /* Build enable-bits array for built-ins. */
63
+ int *bits = (int *)malloc((size_t)NUM_PATTERNS * sizeof(int));
64
+ if (!bits) rb_raise(rb_eNoMemError, "enable_bits allocation failed");
65
+ long alen = RARRAY_LEN(rb_enable_bits);
66
+ for (int i = 0; i < NUM_PATTERNS; i++) {
67
+ if (i < alen) {
68
+ VALUE v = rb_ary_entry(rb_enable_bits, i);
69
+ bits[i] = (RTEST(v) && NUM2INT(v) != 0) ? 1 : 0;
70
+ } else {
71
+ bits[i] = 0;
72
+ }
73
+ }
36
74
 
37
- char *working = strdup(input);
38
- if (!working) rb_raise(rb_eNoMemError, "strdup failed");
75
+ /* Scan + resolve, growing buffer if needed. */
76
+ size_t cap = in_len / 4 + 16;
77
+ mm_match_t *ev = NULL;
78
+ size_t n_ev;
79
+ for (;;) {
80
+ mm_match_t *grown = (mm_match_t *)realloc(ev, cap * sizeof(mm_match_t));
81
+ if (!grown) { free(ev); free(bits); rb_raise(rb_eNoMemError, "mm_scan alloc"); }
82
+ ev = grown;
83
+ n_ev = mm_scan(input, in_len, bits, (size_t)NUM_PATTERNS, ev, cap);
84
+ if (n_ev < cap) break;
85
+ cap *= 2;
86
+ }
87
+ free(bits);
88
+ n_ev = mm_resolve(ev, n_ev);
39
89
 
90
+ /* Collect built-in match hashes. */
40
91
  VALUE matches_arr = rb_ary_new();
92
+ for (size_t i = 0; i < n_ev; i++) {
93
+ int pid = ev[i].pattern_id;
94
+ VALUE h = rb_hash_new();
95
+ rb_hash_aset(h, ID2SYM(rb_intern("tag")),
96
+ ID2SYM(rb_intern(tag_name_for_bit(pattern_tags[pid]))));
97
+ rb_hash_aset(h, ID2SYM(rb_intern("name")),
98
+ rb_str_new_cstr(pattern_names[pid]));
99
+ rb_hash_aset(h, ID2SYM(rb_intern("value")),
100
+ rb_str_new(input + ev[i].start, ev[i].length));
101
+ rb_hash_aset(h, ID2SYM(rb_intern("start")),
102
+ LONG2NUM((long)ev[i].start));
103
+ rb_hash_aset(h, ID2SYM(rb_intern("length")),
104
+ LONG2NUM((long)ev[i].length));
105
+ rb_ary_push(matches_arr, h);
106
+ }
41
107
 
42
- typedef struct { long wpos; long orig_len; } repl_t;
43
- repl_t *repl_log = NULL;
44
- int repl_count = 0;
45
- int repl_cap = 0;
46
-
47
- #define REPL_LOG_PUSH(_wpos, _olen) do { \
48
- if (repl_count >= repl_cap) { \
49
- int _nc = repl_cap == 0 ? 16 : repl_cap * 2; \
50
- repl_t *_t = (repl_t *)realloc(repl_log, sizeof(repl_t) * _nc); \
51
- if (!_t) { free(repl_log); free(working); rb_raise(rb_eNoMemError, "repl_log"); } \
52
- repl_log = _t; repl_cap = _nc; \
53
- } \
54
- repl_log[repl_count].wpos = (_wpos); \
55
- repl_log[repl_count].orig_len = (_olen); \
56
- repl_count++; \
57
- } while (0)
58
-
59
- #define WORKING_TO_ORIG(_wpos) ({ \
60
- long _shift = 0; \
61
- for (int _ri = 0; _ri < repl_count; _ri++) { \
62
- if (repl_log[_ri].wpos <= (_wpos)) \
63
- _shift += 10 - repl_log[_ri].orig_len; \
64
- } \
65
- (_wpos) - _shift; \
66
- })
67
-
68
- #define COLLECT_AND_REPLACE(pat, use_bnd, tag_bit, pat_name) do { \
69
- const char *_cur = working; \
70
- regmatch_t _m[4]; \
71
- while (regexec((pat), _cur, 4, _m, 0) == 0) { \
72
- regoff_t _fso = _m[0].rm_so, _feo = _m[0].rm_eo; \
73
- if (_fso < 0 || _feo < _fso) break; \
74
- regoff_t _cso = _fso, _ceo = _feo; \
75
- if (use_bnd) { \
76
- if (_m[1].rm_so >= 0 && _m[1].rm_eo > _m[1].rm_so) \
77
- _cso = _m[1].rm_eo; \
78
- if (_m[3].rm_so >= 0 && _m[3].rm_eo > _m[3].rm_so) \
79
- _ceo = _m[3].rm_so; \
80
- } \
81
- size_t _vlen = (size_t)(_ceo - _cso); \
82
- long _wpos = (long)(_cur - working) + (long)_cso; \
83
- long _orig = WORKING_TO_ORIG(_wpos); \
84
- VALUE _match = rb_hash_new(); \
85
- rb_hash_aset(_match, ID2SYM(rb_intern("tag")), \
86
- ID2SYM(rb_intern(tag_name_for_bit(tag_bit)))); \
87
- rb_hash_aset(_match, ID2SYM(rb_intern("name")), \
88
- rb_str_new_cstr(pat_name)); \
89
- rb_hash_aset(_match, ID2SYM(rb_intern("value")), \
90
- rb_str_new(_cur + _cso, _vlen)); \
91
- rb_hash_aset(_match, ID2SYM(rb_intern("start")), \
92
- LONG2NUM(_orig)); \
93
- rb_hash_aset(_match, ID2SYM(rb_intern("length")), \
94
- LONG2NUM((long)_vlen)); \
95
- rb_ary_push(matches_arr, _match); \
96
- REPL_LOG_PUSH(_wpos, (long)_vlen); \
97
- if (_feo == _fso) { if (*_cur) _cur++; else break; } \
98
- else _cur += _feo; \
99
- } \
100
- char *_next = replace_all_matches((pat), working, (use_bnd), &ph_default); \
101
- free(working); \
102
- if (!_next) { free(repl_log); rb_raise(rb_eNoMemError, "replace_all_matches failed in scan"); } \
103
- working = _next; \
104
- } while (0)
108
+ /* Build the redacted working buffer (same logic as redact_builtins). */
109
+ size_t ph_len = strlen(ph_plain.str); /* "[REDACTED]" = 10 */
110
+ size_t out_cap = in_len + n_ev * ph_len + 1;
111
+ char *working = (char *)malloc(out_cap);
112
+ if (!working) { free(ev); rb_raise(rb_eNoMemError, "scan working buffer alloc"); }
105
113
 
106
- for (int i = 0; i < NUM_PATTERNS; i++) {
107
- if (!scan_enable_bit(rb_enable_bits, i)) continue;
108
- COLLECT_AND_REPLACE(&compiled_patterns[i], boundary_wrapped[i],
109
- pattern_tags[i], pattern_names[i]);
114
+ size_t out_len = 0, cur = 0;
115
+ for (size_t i = 0; i < n_ev; i++) {
116
+ size_t s = ev[i].start, l = ev[i].length;
117
+ if (s > cur) { memcpy(working + out_len, input + cur, s - cur); out_len += s - cur; }
118
+ memcpy(working + out_len, ph_plain.str, ph_len);
119
+ out_len += ph_len;
120
+ cur = s + l;
110
121
  }
122
+ if (cur < in_len) { memcpy(working + out_len, input + cur, in_len - cur); out_len += in_len - cur; }
123
+ working[out_len] = '\0';
111
124
 
125
+ /* ------------------------------------------------------------------ */
126
+ /* Stage 2: custom patterns via glibc on the rewritten buffer. */
127
+ /* Original coords recovered via working_to_orig() using ev[]. */
128
+ /* ------------------------------------------------------------------ */
112
129
  for (int i = 0; i < custom_count; i++) {
113
130
  if (!scan_enable_bit(rb_enable_bits, NUM_PATTERNS + i)) continue;
114
- COLLECT_AND_REPLACE(&custom_patterns[i].compiled,
115
- custom_patterns[i].boundary,
116
- custom_patterns[i].tag, custom_patterns[i].name);
117
- }
118
131
 
119
- #undef COLLECT_AND_REPLACE
120
- #undef WORKING_TO_ORIG
121
- #undef REPL_LOG_PUSH
132
+ const char *cur_ptr = working;
133
+ regmatch_t m[4];
134
+ while (regexec(&custom_patterns[i].compiled, cur_ptr, 4, m, 0) == 0) {
135
+ regoff_t fso = m[0].rm_so, feo = m[0].rm_eo;
136
+ if (fso < 0 || feo < fso) break;
137
+
138
+ regoff_t cso = fso, ceo = feo;
139
+ if (custom_patterns[i].boundary) {
140
+ if (m[1].rm_so >= 0 && m[1].rm_eo > m[1].rm_so) cso = m[1].rm_eo;
141
+ if (m[3].rm_so >= 0 && m[3].rm_eo > m[3].rm_so) ceo = m[3].rm_so;
142
+ }
143
+
144
+ long wpos_core = (long)(cur_ptr - working) + (long)cso;
145
+ long orig_start = working_to_orig(wpos_core, ev, n_ev, ph_len);
146
+ long core_len = (long)(ceo - cso);
147
+
148
+ VALUE h = rb_hash_new();
149
+ rb_hash_aset(h, ID2SYM(rb_intern("tag")),
150
+ ID2SYM(rb_intern(tag_name_for_bit(custom_patterns[i].tag))));
151
+ rb_hash_aset(h, ID2SYM(rb_intern("name")),
152
+ rb_str_new_cstr(custom_patterns[i].name));
153
+ rb_hash_aset(h, ID2SYM(rb_intern("value")),
154
+ rb_str_new(cur_ptr + cso, (size_t)core_len));
155
+ rb_hash_aset(h, ID2SYM(rb_intern("start")), LONG2NUM(orig_start));
156
+ rb_hash_aset(h, ID2SYM(rb_intern("length")), LONG2NUM(core_len));
157
+ rb_ary_push(matches_arr, h);
158
+
159
+ if (feo == fso) { if (*cur_ptr) cur_ptr++; else break; }
160
+ else cur_ptr += feo;
161
+ }
162
+
163
+ char *next = replace_all_matches(&custom_patterns[i].compiled, working,
164
+ custom_patterns[i].boundary, &ph_plain);
165
+ free(working);
166
+ if (!next) { free(ev); rb_raise(rb_eNoMemError, "replace_all_matches failed in scan"); }
167
+ working = next;
168
+ }
122
169
 
123
- free(repl_log);
170
+ free(ev);
124
171
 
125
- VALUE result = rb_hash_new();
172
+ VALUE result = rb_hash_new();
126
173
  VALUE rb_redacted = rb_str_new_cstr(working);
127
174
  free(working);
175
+ rb_funcall(rb_redacted, rb_intern("force_encoding"), 1,
176
+ rb_funcall(rb_text, rb_intern("encoding"), 0));
128
177
  rb_hash_aset(result, ID2SYM(rb_intern("redacted")), rb_redacted);
129
178
  rb_hash_aset(result, ID2SYM(rb_intern("matches")), matches_arr);
130
179
  return result;
@@ -1,4 +1,4 @@
1
1
  module DataRedactor
2
2
  # Current gem version. Follows {https://semver.org Semantic Versioning 2.0.0}.
3
- VERSION = "0.9.0"
3
+ VERSION = "0.10.1"
4
4
  end
data/lib/data_redactor.rb CHANGED
@@ -74,6 +74,15 @@ module DataRedactor
74
74
  # Default placeholder used when +placeholder:+ is not given to {redact}.
75
75
  PLACEHOLDER_DEFAULT = "[REDACTED]"
76
76
 
77
+ # @api private
78
+ # Inputs larger than this (bytes) are split into newline-bounded chunks before
79
+ # being handed to the C engine. Bounds the per-call O(N) cost glibc regexec
80
+ # pays for state-log allocation, turning total redaction cost from O(N²) (one
81
+ # giant pass) into O(N × CHUNK_SIZE) (many bounded passes). 64 KB is a
82
+ # compromise: small enough to keep per-call cost low, large enough that
83
+ # typical log/JSON inputs use few chunks. See option G in TODO.md.
84
+ CHUNK_SIZE = 64 * 1024
85
+
77
86
  module_function
78
87
 
79
88
  # List of supported tag symbols.
@@ -132,6 +141,11 @@ module DataRedactor
132
141
  def redact(text, only: nil, except: nil, placeholder: PLACEHOLDER_DEFAULT)
133
142
  enable_bits = build_enable_bits(only, except)
134
143
  ph_mode, ph_str = resolve_placeholder(placeholder)
144
+ # Defer to the C layer's TypeError for non-Strings; only chunk if the input
145
+ # is a String big enough to benefit (avoid bytesize on non-Strings).
146
+ if text.is_a?(String) && text.bytesize > CHUNK_SIZE
147
+ return _chunk_bytes(text).map { |c| _redact(c, ph_mode, ph_str, enable_bits) }.join
148
+ end
135
149
  _redact(text, ph_mode, ph_str, enable_bits)
136
150
  end
137
151
 
@@ -157,7 +171,12 @@ module DataRedactor
157
171
  # # value: "user@example.com", start: 0, length: 16}] }
158
172
  def scan(text, only: nil, except: nil)
159
173
  enable_bits = build_enable_bits(only, except)
160
- result = _scan(text, enable_bits)
174
+ result =
175
+ if text.is_a?(String) && text.bytesize > CHUNK_SIZE
176
+ _chunked_scan(text, enable_bits)
177
+ else
178
+ _scan(text, enable_bits)
179
+ end
161
180
  # Normalise: convert tag string from C (uppercase) back to the Symbol used in TAGS
162
181
  result[:matches].each { |m| m[:tag] = m[:tag].to_s.downcase.to_sym }
163
182
  result
@@ -419,4 +438,59 @@ module DataRedactor
419
438
  "placeholder must be a String, :tagged, or :hash — got #{placeholder.inspect}"
420
439
  end
421
440
  end
441
+
442
+ # @api private
443
+ # Split +text+ into byte-bounded chunks for the chunked redact/scan path.
444
+ # Chunks end at a +\n+ when possible so no match straddles a boundary; if a
445
+ # single line exceeds {CHUNK_SIZE} (rare in real inputs), it becomes one
446
+ # oversized chunk and pays the per-pattern O(N) cost — documented limitation.
447
+ # Returns an Array of byte-Strings whose concatenation equals +text+ exactly
448
+ # (including the original newline separators).
449
+ #
450
+ # @param text [String]
451
+ # @return [Array<String>]
452
+ def _chunk_bytes(text)
453
+ chunks = []
454
+ pos = 0
455
+ len = text.bytesize
456
+ while pos < len
457
+ remaining = len - pos
458
+ if remaining <= CHUNK_SIZE
459
+ chunks << text.byteslice(pos, remaining)
460
+ break
461
+ end
462
+ # Find the last \n in [pos, pos+CHUNK_SIZE). If none, chunk is one long
463
+ # line — take CHUNK_SIZE bytes as a fallback (boundary-split risk).
464
+ window = text.byteslice(pos, CHUNK_SIZE)
465
+ nl = window.rindex("\n")
466
+ take = nl ? nl + 1 : CHUNK_SIZE
467
+ chunks << text.byteslice(pos, take)
468
+ pos += take
469
+ end
470
+ chunks
471
+ end
472
+
473
+ # @api private
474
+ # Chunked variant of +_scan+: runs the C scanner on each chunk, then offsets
475
+ # each match's +:start+ by the chunk's base byte-position in the original
476
+ # input so the byteslice invariant holds end-to-end.
477
+ #
478
+ # @param text [String]
479
+ # @param enable_bits [Array<Integer>]
480
+ # @return [Hash{Symbol => Object}] +{ redacted: String, matches: Array<Hash> }+
481
+ def _chunked_scan(text, enable_bits)
482
+ redacted = +""
483
+ matches = []
484
+ base = 0
485
+ _chunk_bytes(text).each do |chunk|
486
+ part = _scan(chunk, enable_bits)
487
+ redacted << part[:redacted]
488
+ part[:matches].each do |m|
489
+ m[:start] += base
490
+ matches << m
491
+ end
492
+ base += chunk.bytesize
493
+ end
494
+ { redacted: redacted, matches: matches }
495
+ end
422
496
  end
metadata CHANGED
@@ -1,7 +1,7 @@
1
1
  --- !ruby/object:Gem::Specification
2
2
  name: data_redactor
3
3
  version: !ruby/object:Gem::Version
4
- version: 0.9.0
4
+ version: 0.10.1
5
5
  platform: ruby
6
6
  authors:
7
7
  - Daniele Frisanco
@@ -79,6 +79,34 @@ dependencies:
79
79
  - - ">="
80
80
  - !ruby/object:Gem::Version
81
81
  version: '2.0'
82
+ - !ruby/object:Gem::Dependency
83
+ name: benchmark-ips
84
+ requirement: !ruby/object:Gem::Requirement
85
+ requirements:
86
+ - - "~>"
87
+ - !ruby/object:Gem::Version
88
+ version: '2.13'
89
+ type: :development
90
+ prerelease: false
91
+ version_requirements: !ruby/object:Gem::Requirement
92
+ requirements:
93
+ - - "~>"
94
+ - !ruby/object:Gem::Version
95
+ version: '2.13'
96
+ - !ruby/object:Gem::Dependency
97
+ name: benchmark-memory
98
+ requirement: !ruby/object:Gem::Requirement
99
+ requirements:
100
+ - - "~>"
101
+ - !ruby/object:Gem::Version
102
+ version: '0.2'
103
+ type: :development
104
+ prerelease: false
105
+ version_requirements: !ruby/object:Gem::Requirement
106
+ requirements:
107
+ - - "~>"
108
+ - !ruby/object:Gem::Version
109
+ version: '0.2'
82
110
  description: A Ruby gem with a C extension for high-performance scanning and redaction
83
111
  of 85 sensitive patterns — API keys, tokens, credentials, IBANs, national IDs, emails,
84
112
  phone numbers, and PII from 15+ countries. Optional Logger formatter, Rails filter_parameters
@@ -93,10 +121,13 @@ extra_rdoc_files: []
93
121
  files:
94
122
  - CHANGELOG.md
95
123
  - LICENSE
124
+ - README.md
96
125
  - ext/data_redactor/custom_patterns.c
97
126
  - ext/data_redactor/custom_patterns.h
98
127
  - ext/data_redactor/data_redactor.c
99
128
  - ext/data_redactor/extconf.rb
129
+ - ext/data_redactor/matcher.c
130
+ - ext/data_redactor/matcher.h
100
131
  - ext/data_redactor/patterns.c
101
132
  - ext/data_redactor/patterns.h
102
133
  - ext/data_redactor/placeholder.c
@@ -112,7 +143,6 @@ files:
112
143
  - lib/data_redactor/integrations/rails.rb
113
144
  - lib/data_redactor/name_pattern.rb
114
145
  - lib/data_redactor/version.rb
115
- - readme.md
116
146
  homepage: https://github.com/danielefrisanco/data_redactor
117
147
  licenses:
118
148
  - MIT