iriq 0.33.0 → 0.35.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +93 -0
- data/README.md +187 -50
- data/lib/iriq/cli.rb +131 -60
- data/lib/iriq/cluster.rb +15 -9
- data/lib/iriq/corpus.rb +233 -117
- data/lib/iriq/errors.rb +9 -0
- data/lib/iriq/normalizer.rb +21 -18
- data/lib/iriq/observation.rb +11 -6
- data/lib/iriq/position_evidence.rb +31 -0
- data/lib/iriq/position_stats.rb +4 -2
- data/lib/iriq/recognizer_proposal.rb +13 -1
- data/lib/iriq/reducer.rb +1 -2
- data/lib/iriq/segment_classifier.rb +9 -1
- data/lib/iriq/storage/json.rb +21 -7
- data/lib/iriq/storage/memory.rb +52 -10
- data/lib/iriq/storage/sqlite.rb +291 -41
- data/lib/iriq/storage.rb +18 -0
- data/lib/iriq/trace.rb +29 -33
- data/lib/iriq/version.rb +1 -1
- data/lib/iriq.rb +1 -0
- metadata +2 -1
|
@@ -130,7 +130,7 @@ module Iriq
|
|
|
130
130
|
|
|
131
131
|
RecognizerProposal.new(
|
|
132
132
|
prefix: prefix,
|
|
133
|
-
suggested_type: prefix
|
|
133
|
+
suggested_type: suggested_type_for(prefix),
|
|
134
134
|
positions: acc[:positions].to_a,
|
|
135
135
|
hosts: acc[:hosts],
|
|
136
136
|
coverage: coverage,
|
|
@@ -144,8 +144,20 @@ module Iriq
|
|
|
144
144
|
}.sort_by { |p| [-p.confidence, p.prefix] }
|
|
145
145
|
end
|
|
146
146
|
|
|
147
|
+
# Placeholder names iriq already means something by. `ip` is the
|
|
148
|
+
# ipv4/ipv6 display umbrella, not a classifier type.
|
|
149
|
+
RESERVED_TYPE_NAMES = (SegmentClassifier::TYPES + %i[ip]).freeze
|
|
150
|
+
|
|
147
151
|
private
|
|
148
152
|
|
|
153
|
+
# A proposal never takes a built-in name — activating `literal_` as
|
|
154
|
+
# :literal would turn every matching value into a fixed literal — so
|
|
155
|
+
# those get an `_id` suffix (`literal_` → :literal_id).
|
|
156
|
+
def suggested_type_for(prefix)
|
|
157
|
+
name = prefix.chomp("_").to_sym
|
|
158
|
+
RESERVED_TYPE_NAMES.include?(name) ? :"#{name}_id" : name
|
|
159
|
+
end
|
|
160
|
+
|
|
149
161
|
def empty_accumulator
|
|
150
162
|
{
|
|
151
163
|
positions: Set.new,
|
data/lib/iriq/reducer.rb
CHANGED
|
@@ -8,8 +8,7 @@ module Iriq
|
|
|
8
8
|
# handles it, register it in DEFAULTS — no other module changes.
|
|
9
9
|
module Reducer
|
|
10
10
|
# Each entry: { event_class => [lambda(event, storage) -> result] }.
|
|
11
|
-
# Lambdas
|
|
12
|
-
# callers (Corpus#observe) can pick up the cluster they need to return.
|
|
11
|
+
# Lambdas return the result of the underlying storage call.
|
|
13
12
|
DEFAULTS = {
|
|
14
13
|
Event::HostSeen => [->(e, s) { s.increment_host(e.host) }],
|
|
15
14
|
Event::PathLengthSeen => [->(e, s) { s.increment_path_length(e.length) }],
|
|
@@ -233,6 +233,13 @@ module Iriq
|
|
|
233
233
|
recognizer
|
|
234
234
|
end
|
|
235
235
|
|
|
236
|
+
# A copy registers recognizers without touching the original.
|
|
237
|
+
def initialize_copy(source)
|
|
238
|
+
super
|
|
239
|
+
@recognizers = source.recognizers
|
|
240
|
+
@cache = {}
|
|
241
|
+
end
|
|
242
|
+
|
|
236
243
|
# Snapshot of the live ensemble. Useful for tests and tooling that
|
|
237
244
|
# want to inspect which Recognizers a corpus is consulting.
|
|
238
245
|
def recognizers
|
|
@@ -505,7 +512,8 @@ module Iriq
|
|
|
505
512
|
# /pricing/USD both render as /pricing/USD.
|
|
506
513
|
def self.canonical_currency(value)
|
|
507
514
|
return nil if value.nil?
|
|
508
|
-
|
|
515
|
+
# ASCII-only: full Unicode upcase maps ſ→S / ı→I, forging a code.
|
|
516
|
+
up = value.upcase(:ascii)
|
|
509
517
|
CURRENCY_CODES.include?(up) ? up : nil
|
|
510
518
|
end
|
|
511
519
|
|
data/lib/iriq/storage/json.rb
CHANGED
|
@@ -19,24 +19,38 @@ module Iriq
|
|
|
19
19
|
s
|
|
20
20
|
end
|
|
21
21
|
|
|
22
|
+
# Refuses (Iriq::CorpusError) anything that isn't a corpus dump, so a
|
|
23
|
+
# mistyped --corpus path can't overwrite an unrelated JSON file: it must
|
|
24
|
+
# be an object with at least one corpus key, or `{}`.
|
|
22
25
|
def load!(path)
|
|
23
|
-
data =
|
|
26
|
+
data = begin
|
|
27
|
+
File.read(path)
|
|
28
|
+
rescue SystemCallError => e
|
|
29
|
+
raise CorpusError, "corpus #{path}: #{Iriq.os_error_message(e)}"
|
|
30
|
+
end
|
|
24
31
|
return self if data.empty?
|
|
25
32
|
|
|
26
|
-
|
|
33
|
+
dump = begin
|
|
34
|
+
JSON.parse(data)
|
|
35
|
+
rescue JSON::ParserError
|
|
36
|
+
raise CorpusError, "corpus #{path}: not valid JSON"
|
|
37
|
+
end
|
|
38
|
+
unless dump.is_a?(Hash) && (dump.empty? || dump.keys.intersect?(DUMP_KEYS))
|
|
39
|
+
raise CorpusError, "corpus #{path}: not an iriq corpus (no corpus keys at the top level)"
|
|
40
|
+
end
|
|
41
|
+
|
|
42
|
+
load_dump!(dump)
|
|
27
43
|
@path = path
|
|
28
44
|
self
|
|
29
45
|
end
|
|
30
46
|
|
|
31
|
-
# save writes atomically (tmp + rename). Defaults to the path
|
|
32
|
-
# open(); pass an explicit path to write elsewhere.
|
|
47
|
+
# save writes atomically (unique tmp + rename). Defaults to the path
|
|
48
|
+
# passed at open(); pass an explicit path to write elsewhere.
|
|
33
49
|
def save(path = nil)
|
|
34
50
|
target = path || @path
|
|
35
51
|
raise ArgumentError, "no path provided" unless target
|
|
36
52
|
|
|
37
|
-
|
|
38
|
-
File.write(tmp, JSON.generate(to_dump))
|
|
39
|
-
File.rename(tmp, target)
|
|
53
|
+
Storage.write_atomically(target, JSON.generate(to_dump))
|
|
40
54
|
end
|
|
41
55
|
end
|
|
42
56
|
end
|
data/lib/iriq/storage/memory.rb
CHANGED
|
@@ -15,12 +15,17 @@ module Iriq
|
|
|
15
15
|
#
|
|
16
16
|
# host_counts / path_length_counts / raw_shape_counts / fingerprint_counts
|
|
17
17
|
# position_stats(position)
|
|
18
|
+
# position_evidence(position, value) # the narrow read normalize uses
|
|
19
|
+
# param_stats(cluster_key, name) # one param, without the cluster
|
|
18
20
|
# each_position_stats { |position, stats| ... }
|
|
19
21
|
# each_observed_iri { |canonical| ... }
|
|
20
|
-
#
|
|
22
|
+
# each_observed_iri_since(mark) { |canonical| ... } # → mark of the last
|
|
23
|
+
# clear_materialized_views
|
|
24
|
+
# begin_rebuild / install_rebuild / discard_rebuild # for reinfer
|
|
21
25
|
# clusters / cluster_size
|
|
22
26
|
#
|
|
23
27
|
# transaction { ... } # backends may batch within
|
|
28
|
+
# turn_over? # a long batch should commit and let others in
|
|
24
29
|
# flush # commit pending writes (no-op for Memory)
|
|
25
30
|
# close # release resources
|
|
26
31
|
class Memory
|
|
@@ -55,8 +60,14 @@ module Iriq
|
|
|
55
60
|
yield self
|
|
56
61
|
end
|
|
57
62
|
|
|
63
|
+
# Yields false: nothing else writes an in-memory corpus.
|
|
58
64
|
def batch
|
|
59
|
-
yield
|
|
65
|
+
yield false
|
|
66
|
+
end
|
|
67
|
+
|
|
68
|
+
# Nothing else waits on an in-memory corpus.
|
|
69
|
+
def turn_over?
|
|
70
|
+
false
|
|
60
71
|
end
|
|
61
72
|
|
|
62
73
|
def flush; end
|
|
@@ -108,6 +119,13 @@ module Iriq
|
|
|
108
119
|
@observed_iris.each(&block)
|
|
109
120
|
end
|
|
110
121
|
|
|
122
|
+
# The observations logged after `mark` (0 for all), in order; returns
|
|
123
|
+
# the mark of the last one.
|
|
124
|
+
def each_observed_iri_since(mark, &block)
|
|
125
|
+
@observed_iris.drop(mark).each(&block)
|
|
126
|
+
[mark, @observed_iris.size].max
|
|
127
|
+
end
|
|
128
|
+
|
|
111
129
|
def observed_iri_count
|
|
112
130
|
@observed_iris.size
|
|
113
131
|
end
|
|
@@ -138,6 +156,14 @@ module Iriq
|
|
|
138
156
|
@clusters = {}
|
|
139
157
|
end
|
|
140
158
|
|
|
159
|
+
# Nothing else reads an in-memory corpus mid-rebuild: rebuild in place.
|
|
160
|
+
def begin_rebuild
|
|
161
|
+
clear_materialized_views
|
|
162
|
+
end
|
|
163
|
+
|
|
164
|
+
def install_rebuild; end
|
|
165
|
+
def discard_rebuild; end
|
|
166
|
+
|
|
141
167
|
# --- Reads ------------------------------------------------------------
|
|
142
168
|
|
|
143
169
|
def host_counts; @host_counts; end
|
|
@@ -149,6 +175,12 @@ module Iriq
|
|
|
149
175
|
@position_stats[position]
|
|
150
176
|
end
|
|
151
177
|
|
|
178
|
+
# Built over the live stats, so nothing is copied.
|
|
179
|
+
def position_evidence(position, value)
|
|
180
|
+
stats = @position_stats[position]
|
|
181
|
+
stats && PositionEvidence.from_stats(stats, value)
|
|
182
|
+
end
|
|
183
|
+
|
|
152
184
|
def each_position_stats(&block)
|
|
153
185
|
@position_stats.each(&block)
|
|
154
186
|
end
|
|
@@ -161,22 +193,32 @@ module Iriq
|
|
|
161
193
|
@clusters.size
|
|
162
194
|
end
|
|
163
195
|
|
|
164
|
-
# O(1) lookup by cluster key
|
|
165
|
-
#
|
|
166
|
-
# has been observed under this key yet.
|
|
196
|
+
# O(1) lookup by cluster key. nil if no cluster has been observed under
|
|
197
|
+
# this key yet.
|
|
167
198
|
def cluster_for(key)
|
|
168
199
|
@clusters[key]
|
|
169
200
|
end
|
|
170
201
|
|
|
202
|
+
def param_stats(key, name)
|
|
203
|
+
cluster = @clusters[key]
|
|
204
|
+
cluster && cluster.param_stats[name]
|
|
205
|
+
end
|
|
206
|
+
|
|
171
207
|
# --- Bulk load (used by JSON backend) --------------------------------
|
|
172
208
|
|
|
209
|
+
# Top-level keys of the dump. A JSON file with none of them isn't a corpus.
|
|
210
|
+
DUMP_KEYS = %w[host_counts path_length_counts raw_shape_counts fingerprint_counts
|
|
211
|
+
max_values_per_position position_stats clusterer observed_iris
|
|
212
|
+
activated_recognizers].freeze
|
|
213
|
+
|
|
214
|
+
# Missing keys load as empty, so `{}` is a valid (empty) corpus.
|
|
173
215
|
def load_dump!(h)
|
|
174
|
-
@host_counts = Hash.new(0).merge(h
|
|
175
|
-
@path_length_counts = Hash.new(0).merge(h
|
|
176
|
-
@raw_shape_counts = Hash.new(0).merge(h
|
|
177
|
-
@fingerprint_counts = Hash.new(0).merge(h
|
|
216
|
+
@host_counts = Hash.new(0).merge(h.fetch("host_counts", {}))
|
|
217
|
+
@path_length_counts = Hash.new(0).merge(h.fetch("path_length_counts", {}).transform_keys(&:to_i))
|
|
218
|
+
@raw_shape_counts = Hash.new(0).merge(h.fetch("raw_shape_counts", {}))
|
|
219
|
+
@fingerprint_counts = Hash.new(0).merge(h.fetch("fingerprint_counts", {}))
|
|
178
220
|
@max_values_per_position = h.fetch("max_values_per_position", PositionStats::DEFAULT_MAX_VALUES)
|
|
179
|
-
@position_stats = h
|
|
221
|
+
@position_stats = h.fetch("position_stats", []).each_with_object({}) do |entry, acc|
|
|
180
222
|
position = Position.from_dump(entry["position"])
|
|
181
223
|
acc[position] = PositionStats.from_dump(entry["stats"])
|
|
182
224
|
end
|