gemchat 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,555 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "sqlite3"
4
+
5
+ module Gemchat
6
+ # Local index store: SQLite with FTS5 for BM25. No server, no extension.
7
+ #
8
+ # Vectors are not stored here yet; the schema reserves `embedding` and
9
+ # `embed_model` so §7.1's store identity can refuse to open a store written by
10
+ # a different model rather than silently returning wrong neighbours.
11
+ class Store
12
+ SCHEMA_VERSION = "2"
13
+ BACKEND = "local"
14
+
15
+ # Fields that decide whether a store is readable at all (§7.1).
16
+ #
17
+ # `schema_version` is the hard gate: a newer or older layout cannot be
18
+ # queried correctly, and silently opening it produces wrong answers rather
19
+ # than an error. `gemchat_version` is recorded but tolerated — a point release
20
+ # that only adds a command must not strand a user's index. `embedding_model`
21
+ # and `embedding_dimensions` are placeholders for the embedding phase, where
22
+ # mixing vectors from different models *is* silently wrong.
23
+ IDENTITY_FIELDS = %w[schema_version gemchat_version embedding_model
24
+ embedding_dimensions backend].freeze
25
+
26
+ class IdentityError < Gemchat::Error; end
27
+
28
+ def self.default_path
29
+ File.join(Gemchat.root, "index.sqlite")
30
+ end
31
+
32
+ def initialize(path = self.class.default_path)
33
+ @path = path
34
+ @drift = []
35
+ FileUtils.mkdir_p(File.dirname(path))
36
+ @db = SQLite3::Database.new(path)
37
+ @db.results_as_hash = true
38
+ register_vector_functions
39
+ # Verified before migrating. migrate! rewrites schema_version on every
40
+ # open, so checking afterwards would compare the value against itself and
41
+ # always agree.
42
+ verify_identity!
43
+ migrate!
44
+ record_identity!
45
+ end
46
+
47
+ attr_reader :path
48
+
49
+ # Non-fatal differences worth surfacing in `status`.
50
+ attr_reader :drift
51
+
52
+ def verify_identity!
53
+ return unless table?("meta")
54
+
55
+ stored = meta("schema_version")
56
+ return if stored.nil? || stored == SCHEMA_VERSION
57
+
58
+ raise IdentityError, <<~MSG.strip
59
+ this index was written by a different gemchat (store schema #{stored}, this build expects #{SCHEMA_VERSION}).
60
+ Refusing to read it rather than returning wrong results.
61
+ Rebuild it with: gemchat reindex
62
+ MSG
63
+ end
64
+
65
+ def record_identity!
66
+ # Read before writing. Setting the field first and comparing after would
67
+ # compare the new value against itself and never report drift, which is
68
+ # the one thing this is here to notice.
69
+ previous_version = meta("gemchat_version")
70
+
71
+ set_meta("schema_version", SCHEMA_VERSION)
72
+ set_meta("gemchat_version", Gemchat::VERSION)
73
+ set_meta("backend", BACKEND)
74
+
75
+ @drift << "gemchat_version" if previous_version && previous_version != Gemchat::VERSION
76
+ drift
77
+ end
78
+
79
+ # SQLite has no dot product. This one computes it in C over the float32 blob,
80
+ # which matters: pulling every candidate vector into Ruby to multiply it
81
+ # would be a full scan of the corpus in the interpreter on every query.
82
+ #
83
+ # Dot product equals cosine similarity here because every stored vector is
84
+ # L2-normalised (Rllama::Model#embed normalises by default), so no magnitude
85
+ # term is needed and the result is directly comparable.
86
+ #
87
+ # The result is assigned to the FunctionProxy rather than returned. sqlite3
88
+ # 2.x wraps the block and reads `fp.result`; the block's own return value is
89
+ # discarded. Returning the float instead yields NULL for every row, and
90
+ # because the rank is negated on the way out, that surfaces as a silent 0.0
91
+ # for every result -- correct-looking output, no ordering at all.
92
+ def register_vector_functions
93
+ @db.create_function("dot", 2) do |fp, a, b|
94
+ left = a.to_s.unpack("e*")
95
+ right = b.to_s.unpack("e*")
96
+ fp.result = if left.empty? || left.size != right.size
97
+ nil
98
+ else
99
+ left.zip(right).sum { |x, y| x * y }
100
+ end
101
+ end
102
+ end
103
+
104
+ def table?(name)
105
+ !@db.get_first_value(
106
+ "SELECT 1 FROM sqlite_master WHERE type = 'table' AND name = ?", [name]
107
+ ).nil?
108
+ rescue SQLite3::SQLException
109
+ false
110
+ end
111
+
112
+ def migrate!
113
+ @db.execute_batch(<<~SQL)
114
+ CREATE TABLE IF NOT EXISTS meta (
115
+ key TEXT PRIMARY KEY,
116
+ value TEXT
117
+ );
118
+
119
+ CREATE TABLE IF NOT EXISTS gems (
120
+ name TEXT NOT NULL,
121
+ version TEXT NOT NULL,
122
+ source_type TEXT,
123
+ chunk_count INTEGER NOT NULL DEFAULT 0,
124
+ source_hash TEXT,
125
+ indexed_at TEXT,
126
+ PRIMARY KEY (name, version, source_type)
127
+ );
128
+
129
+ CREATE TABLE IF NOT EXISTS chunks (
130
+ id INTEGER PRIMARY KEY,
131
+ gem_name TEXT NOT NULL,
132
+ gem_version TEXT NOT NULL,
133
+ source_type TEXT NOT NULL,
134
+ title TEXT,
135
+ body TEXT,
136
+ class_name TEXT,
137
+ method_name TEXT,
138
+ method_type TEXT,
139
+ signature TEXT,
140
+ source_path TEXT,
141
+ source_line INTEGER,
142
+ content_digest TEXT NOT NULL,
143
+ embedding BLOB,
144
+ embed_model TEXT,
145
+ UNIQUE (gem_name, gem_version, source_type, class_name, method_name, title)
146
+ );
147
+
148
+ CREATE VIRTUAL TABLE IF NOT EXISTS chunks_fts USING fts5(
149
+ title, body, signature, class_name, method_name,
150
+ content='chunks', content_rowid='id', tokenize='porter unicode61'
151
+ );
152
+ SQL
153
+ end
154
+
155
+ def set_meta(key, value)
156
+ @db.execute("INSERT INTO meta (key, value) VALUES (?, ?) ON CONFLICT(key) DO UPDATE SET value = excluded.value", [key, value.to_s])
157
+ end
158
+
159
+ def meta(key)
160
+ row = @db.get_first_row("SELECT value FROM meta WHERE key = ?", [key])
161
+ row&.fetch("value")
162
+ end
163
+
164
+ def indexed?(name, version, source_type = "ri")
165
+ !@db.get_first_value(
166
+ "SELECT 1 FROM gems WHERE name = ? AND version = ? AND source_type = ?",
167
+ [name, version, source_type]
168
+ ).nil?
169
+ end
170
+
171
+ def gem_summary(name, version, source_type = "ri")
172
+ @db.get_first_row(
173
+ "SELECT * FROM gems WHERE name = ? AND version = ? AND source_type = ?",
174
+ [name, version, source_type]
175
+ )
176
+ end
177
+
178
+ # What a write actually did. "unchanged" is distinct from "indexed" with
179
+ # zero chunks: the first means the digest matched, the second would mean
180
+ # content was lost, and reporting them the same makes a broken run look
181
+ # like a successful no-op.
182
+ Outcome = Struct.new(:status, :chunks) do
183
+ def changed? = status != :unchanged
184
+ end
185
+
186
+ def replace_gem(name, version, source_type, chunks)
187
+ digest = Digest::SHA256.hexdigest(chunks.map { |c| c.content_digest }.join("|"))
188
+ existing = gem_summary(name, version, source_type)
189
+ return Outcome.new(:unchanged, 0) if existing && existing["source_hash"] == digest
190
+ status = existing ? :updated : :indexed
191
+
192
+ @db.transaction do
193
+ old_ids = @db.execute(
194
+ "SELECT id FROM chunks WHERE gem_name = ? AND gem_version = ? AND source_type = ?",
195
+ [name, version, source_type]
196
+ ).map { |r| r["id"] }
197
+ unless old_ids.empty?
198
+ @db.execute("DELETE FROM chunks_fts WHERE rowid IN (#{old_ids.map(&:to_i).join(",")})")
199
+ @db.execute("DELETE FROM chunks WHERE gem_name = ? AND gem_version = ? AND source_type = ?", [name, version, source_type])
200
+ end
201
+
202
+ now = Time.now.utc.iso8601
203
+ chunks.each do |chunk|
204
+ insert_sql = <<~SQL
205
+ INSERT OR IGNORE INTO chunks
206
+ (gem_name, gem_version, source_type, title, body, class_name, method_name,
207
+ method_type, signature, source_path, source_line, content_digest)
208
+ VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
209
+ SQL
210
+ fts_sql = <<~SQL
211
+ INSERT INTO chunks_fts (rowid, title, body, signature, class_name, method_name)
212
+ VALUES (?, ?, ?, ?, ?, ?)
213
+ SQL
214
+ binds = [name, version, source_type, chunk.title, chunk.body,
215
+ chunk.class_name, chunk.method_name, chunk.method_type,
216
+ chunk.signature, chunk.source_path, chunk.source_line,
217
+ chunk.content_digest]
218
+ @db.execute(insert_sql, binds)
219
+ id = @db.last_insert_row_id
220
+ @db.execute(fts_sql, [id, chunk.title, chunk.body, chunk.signature, chunk.class_name, chunk.method_name])
221
+ end
222
+
223
+ @db.execute(<<~SQL, [name, version, source_type, chunks.size, digest, now])
224
+ INSERT INTO gems (name, version, source_type, chunk_count, source_hash, indexed_at)
225
+ VALUES (?, ?, ?, ?, ?, ?)
226
+ ON CONFLICT(name, version, source_type) DO UPDATE SET
227
+ chunk_count = excluded.chunk_count,
228
+ source_hash = excluded.source_hash,
229
+ indexed_at = excluded.indexed_at
230
+ SQL
231
+ end
232
+
233
+ Outcome.new(status, chunks.size)
234
+ end
235
+
236
+ # A gem has one row per source_type, so a plain COUNT(*) would report a gem
237
+ # twice as soon as it also had a README indexed.
238
+ def stats
239
+ {
240
+ gems: @db.get_first_value("SELECT COUNT(DISTINCT name || char(0) || version) FROM gems").to_i,
241
+ chunks: @db.get_first_value("SELECT COUNT(*) FROM chunks").to_i
242
+ }
243
+ end
244
+
245
+ def chunk_counts_by_source
246
+ @db.execute("SELECT source_type, COUNT(*) AS n FROM chunks GROUP BY source_type")
247
+ .to_h { |row| [row["source_type"], row["n"].to_i] }
248
+ end
249
+
250
+ # Explicit column list rather than a splat. `results_as_hash` makes every
251
+ # row carry integer keys too, so `{**row}` would leak 0, 1, 2... into the
252
+ # result and mix symbol and string key styles.
253
+ # `id` is first because fusion needs a stable key: BM25 and vector search
254
+ # return the same chunk as two separate rows, and the identity fields alone
255
+ # are not guaranteed unique. It is not in the CLI's RESULT_KEYS, so it does
256
+ # not leak into `--json` output.
257
+ COLUMNS = %w[id gem_name gem_version source_type title body class_name method_name signature source_path source_line].freeze
258
+
259
+ # Reciprocal rank fusion constant, matching gemchat_app's RRF_K and qmd.
260
+ RRF_K = 60
261
+
262
+ # FTS5 query syntax collides with Ruby naming: an unquoted `Puma::Server`
263
+ # parses as a column filter on `Puma` and fails with "no such column".
264
+ # Quoting each whitespace-separated term makes FTS5 treat it as a phrase and
265
+ # neutralises `:` , `*`, `-`, `"` and friends.
266
+ # The quoting is load-bearing: it is what stops FTS5 reading `Puma::Server` as
267
+ # a column filter, which fails with "no such column" and is a systematic
268
+ # collision with Ruby naming rather than a one-off.
269
+ #
270
+ # But joining the quoted terms with a *space* is FTS5's AND-of-a-phrase, and
271
+ # that was not a decision so much as a side effect of escaping. Measured: on
272
+ # the 236-chunk rake index, a phrase query returns **zero** rows for every
273
+ # natural-language question tried -- "how do I list all the available tasks",
274
+ # "syntax highlighting in the console" -- and a hit only when the query is a
275
+ # literal substring. So `search` was an exact-phrase finder wearing BM25's
276
+ # name, not a keyword search.
277
+ #
278
+ # :phrase the quoted run. Exact substrings, high precision, near-zero recall.
279
+ # The default for `search`, where a caller asking for
280
+ # "Rake::Task#enhance" wants exactly that. Stopwords are never
281
+ # dropped here: removing a word from a literal phrase changes what
282
+ # that phrase matches, which is a different query, not a better one.
283
+ # :any the same terms OR'd, with stopwords dropped. What a keyword search
284
+ # means, and what the fusion's lexical arm needs. See §11.3 for what
285
+ # it measures and §11.4 for the stopword pass and the eval gate.
286
+ def self.fts_query(raw, mode: :phrase)
287
+ terms = raw.to_s.split(/\s+/).reject(&:empty?)
288
+ return nil if terms.empty?
289
+
290
+ if mode == :any
291
+ kept = terms.uniq.reject { |term| Stopwords.only_stopwords?(term) }
292
+ # Every term was a stopword. Fall back to the full set rather than
293
+ # returning nil: the caller asked a real question, and an empty query
294
+ # would report "no results" as though the corpus were empty.
295
+ kept = terms.uniq if kept.empty?
296
+ kept.map { |term| %("#{term.gsub('"', '""')}") }.join(" OR ")
297
+ else
298
+ terms.uniq.map { |term| %("#{term.gsub('"', '""')}") }.join(" ")
299
+ end
300
+ end
301
+
302
+ # BM25's rank() is lower-is-better, so results are exposed as a positive
303
+ # `score` where higher is better. Every key is a String, including "score".
304
+ def search(query, limit: 10, min_score: 0.0, gem_names: nil, mode: :phrase)
305
+ match = self.class.fts_query(query, mode: mode)
306
+ return [] if match.nil?
307
+
308
+ binds = [match]
309
+ scope = ""
310
+ if gem_names && !gem_names.empty?
311
+ scope = "AND c.gem_name IN (#{(["?"] * gem_names.size).join(", ")})"
312
+ binds.concat(gem_names)
313
+ end
314
+ binds << limit
315
+
316
+ sql = <<~SQL
317
+ SELECT #{COLUMNS.map { |c| "c.#{c}" }.join(", ")},
318
+ bm25(chunks_fts) AS rank
319
+ FROM chunks_fts
320
+ JOIN chunks c ON c.id = chunks_fts.rowid
321
+ WHERE chunks_fts MATCH ? #{scope}
322
+ ORDER BY rank
323
+ LIMIT ?
324
+ SQL
325
+
326
+ @db.execute(sql, binds).map { |row|
327
+ COLUMNS.to_h { |c| [c, row[c]] }.merge("score" => -row["rank"].to_f)
328
+ }.select { |row| row["score"] >= min_score }
329
+ end
330
+
331
+ # --- vectors ------------------------------------------------------------
332
+
333
+ # Refuse to write vectors from a model this store has not seen. Two models in
334
+ # one table cannot be compared, and the failure would be silent: cosine
335
+ # between unrelated spaces still returns a number, just a meaningless one.
336
+ # Drops every vector and the metadata that identifies them. Needed because
337
+ # `unembedded_chunks` only returns rows where the vector IS NULL, so a model
338
+ # change cannot be reconciled by re-running `embed` -- without this the
339
+ # mismatch error would tell the user to run a command that does nothing.
340
+ def clear_vectors
341
+ @db.execute("UPDATE chunks SET embedding = NULL WHERE embedding IS NOT NULL")
342
+ dropped = @db.changes
343
+ @db.execute("DELETE FROM meta WHERE key IN ('embed_model', 'embed_dimensions', 'embed_at')")
344
+ dropped
345
+ end
346
+
347
+ def assert_embedding_model!(identity)
348
+ stored = meta("embed_model")
349
+ return true if stored.nil? || stored == identity
350
+
351
+ raise EmbeddingModelMismatch, <<~MSG.strip
352
+ this index already holds vectors from #{stored}, and the current model is #{identity}.
353
+ Vectors from different models cannot be compared, so the store will not mix them.
354
+ Discard the vectors with: gemchat embed --force
355
+ MSG
356
+ end
357
+
358
+ # float32 little-endian: 768 dimensions is 3072 bytes a chunk, so ~43MB for
359
+ # the 14k-chunk corpus the plan measured. Small enough to keep in the row.
360
+ def pack(vector)
361
+ vector.map { |v| v.to_f }.pack("e*")
362
+ end
363
+
364
+ def unpack(blob)
365
+ return [] if blob.nil? || blob.empty?
366
+
367
+ blob.unpack("e*")
368
+ end
369
+
370
+ def store_embeddings(rows, vectors, identity:, dimensions:)
371
+ assert_embedding_model!(identity)
372
+ return 0 if rows.empty?
373
+
374
+ unless vectors.size == rows.size
375
+ raise Error, "embedding count mismatch: #{vectors.size} vectors for #{rows.size} chunks"
376
+ end
377
+
378
+ bad = vectors.find { |v| v.size != dimensions }
379
+ if bad
380
+ raise Error, "expected #{dimensions}-dimension vectors, got #{bad.size}"
381
+ end
382
+
383
+ now = Time.now.utc.iso8601
384
+ @db.transaction do
385
+ rows.each_with_index do |id, i|
386
+ @db.execute("UPDATE chunks SET embedding = ?, embed_model = ? WHERE id = ?",
387
+ [pack(vectors[i]), identity, id])
388
+ end
389
+ end
390
+
391
+ set_meta("embed_model", identity)
392
+ set_meta("embedding_dimensions", dimensions)
393
+ set_meta("embedded_at", now)
394
+ vectors.size
395
+ end
396
+
397
+ # The chunks still needing a vector, with the text to embed. Ordered so a
398
+ # resumed run picks up where it stopped.
399
+ #
400
+ # The text is rebuilt through Chunker::Chunk#searchable_text rather than
401
+ # assembled in SQL. A query embedding and a document embedding must be
402
+ # produced by *identical* text, and a second implementation of that format in
403
+ # SQL would drift from the Ruby one the chunks were written with -- silently,
404
+ # because both still produce a valid-looking vector.
405
+ def unembedded_chunks(limit: nil)
406
+ sql = <<~SQL
407
+ SELECT id, gem_name, gem_version, source_type, title, body,
408
+ class_name, method_name, method_type, signature
409
+ FROM chunks
410
+ WHERE embedding IS NULL
411
+ ORDER BY id
412
+ SQL
413
+ sql += " LIMIT #{limit.to_i}" if limit
414
+
415
+ @db.execute(sql).map { |row|
416
+ chunk = Chunker::Chunk.new(
417
+ gem_name: row["gem_name"], gem_version: row["gem_version"],
418
+ source_type: row["source_type"], title: row["title"], body: row["body"],
419
+ class_name: row["class_name"], method_name: row["method_name"],
420
+ method_type: row["method_type"], signature: row["signature"]
421
+ )
422
+ {"id" => row["id"], "searchable" => chunk.searchable_text}
423
+ }
424
+ end
425
+
426
+ def embedded_stats
427
+ total = @db.get_first_value("SELECT COUNT(*) FROM chunks").to_i
428
+ embedded = @db.get_first_value("SELECT COUNT(*) FROM chunks WHERE embedding IS NOT NULL").to_i
429
+ {total:, embedded:, pending: total - embedded}
430
+ end
431
+
432
+ # Unit vectors, so a dot product is cosine similarity and the ordering falls
433
+ # out of a single pass. Scoped to the same model as the query was embedded
434
+ # with, so a mixed store cannot return a comparison across spaces.
435
+ def vector_search(query_vector, limit: 10, min_score: 0.0, gem_names: nil)
436
+ return [] if query_vector.nil? || query_vector.empty?
437
+
438
+ identity = meta("embed_model")
439
+ return [] if identity.nil?
440
+
441
+ # A mis-sized query makes dot() return NULL for every row, which sorts as
442
+ # 0.0 and would return an arbitrary slice of the corpus as if it were a
443
+ # ranked answer. Refusing up front turns a silent wrong answer into none.
444
+ stored_dimensions = meta("embedding_dimensions")&.to_i
445
+ if stored_dimensions && stored_dimensions != query_vector.size
446
+ return []
447
+ end
448
+
449
+ binds = [pack(query_vector), identity]
450
+ scope = ""
451
+ if gem_names && !gem_names.empty?
452
+ scope = "AND c.gem_name IN (#{(["?"] * gem_names.size).join(", ")})"
453
+ binds.concat(gem_names)
454
+ end
455
+ binds << limit
456
+
457
+ sql = <<~SQL
458
+ SELECT #{COLUMNS.map { |c| "c.#{c}" }.join(", ")},
459
+ -dot(embedding, ?) AS rank
460
+ FROM chunks c
461
+ WHERE c.embedding IS NOT NULL
462
+ AND c.embed_model = ? #{scope}
463
+ ORDER BY rank
464
+ LIMIT ?
465
+ SQL
466
+
467
+ @db.execute(sql, binds).map { |row|
468
+ COLUMNS.to_h { |c| [c, row[c]] }.merge("score" => -row["rank"].to_f)
469
+ }.select { |row| row["score"] >= min_score }
470
+ end
471
+
472
+ # --- hybrid -------------------------------------------------------------
473
+
474
+ # Runs whichever arms were asked for and fuses them by reciprocal rank.
475
+ #
476
+ # The UDF constraint from above does not bite here. It stopped `vector_search`
477
+ # from being narrowed by FTS inside a single statement, but each arm here
478
+ # returns a small ranked list on its own, so the fusion is a hash fold over a
479
+ # few hundred rows in Ruby. That is a different operation from the one that
480
+ # was infeasible, and worth distinguishing before assuming otherwise.
481
+ #
482
+ # RRF is used precisely because the two scores are not comparable: one is a
483
+ # negated BM25 rank on an unbounded scale, the other a cosine in [-1, 1].
484
+ # Fusing on *rank* sidesteps normalisation entirely, which is the reason to
485
+ # reach for it rather than for a weighted sum of the raw scores.
486
+ # `lex_mode` is a parameter rather than a constant because the obvious
487
+ # improvement to :phrase is wrong, and that is worth being able to measure
488
+ # rather than re-derive:
489
+ #
490
+ # :phrase the quoted run. Measured on 22,320 chunks from 34 gems, it
491
+ # returns zero rows for every natural-language question tried, and
492
+ # one exact hit for an identifier. High precision, no recall.
493
+ # :any the same terms OR'd. This is *worse*, not better: on five
494
+ # known-answer queries -- queries whose answer definitely exists in
495
+ # the corpus -- it located the right chunk in 1 of 5. Short ri
496
+ # signatures containing one query word outrank the long prose chunk
497
+ # that actually answers the question, because a rare term in a short
498
+ # document scores enormously under BM25.
499
+ #
500
+ # So :any is not the default. Feeding it to a fusion is actively harmful,
501
+ # because RRF trusts rank position: a confidently wrong arm does not get
502
+ # ignored, it gets promoted. A real keyword arm needs term selection (drop
503
+ # the common words, require a rare term) before it belongs anywhere near
504
+ # :any. See §11.3 of the plan.
505
+ def hybrid_search(query, query_vector: nil, lex: true, lex_mode: :phrase,
506
+ limit: 10, pool: 50, k: RRF_K, gem_names: nil)
507
+ arms = []
508
+
509
+ if lex
510
+ found = search(query, limit: pool, gem_names: gem_names, mode: lex_mode)
511
+ arms << [:bm25, found] unless found.empty?
512
+ end
513
+
514
+ if query_vector
515
+ vec = vector_search(query_vector, limit: pool, gem_names: gem_names)
516
+ arms << [:vector, vec] unless vec.empty?
517
+ end
518
+
519
+ return [] if arms.empty?
520
+
521
+ rrf_fuse(arms, limit: limit, k: k)
522
+ end
523
+
524
+ # `arms` is [[name, ranked_rows], ...]. Each row carries "matched", naming
525
+ # the arms that found it, because two engines agreeing is signal in its own
526
+ # right and the fold would otherwise throw it away.
527
+ def rrf_fuse(arms, limit:, k: RRF_K)
528
+ scores = Hash.new(0.0)
529
+ matched = Hash.new { |h, key| h[key] = [] }
530
+ rows = {}
531
+
532
+ arms.each do |name, list|
533
+ list.each_with_index do |row, i|
534
+ id = row["id"]
535
+ scores[id] += 1.0 / (k + i + 1)
536
+ matched[id] << name
537
+ rows[id] ||= row
538
+ end
539
+ end
540
+
541
+ # Ties break on id so the order is deterministic across runs. Without this,
542
+ # two equally-scored chunks swap places between invocations, which makes a
543
+ # regression look like a change.
544
+ rows.keys.sort_by { |id| [-scores[id], id] }.first(limit).map { |id|
545
+ rows[id].merge("score" => scores[id], "matched" => matched[id])
546
+ }
547
+ end
548
+
549
+ def close
550
+ @db.close
551
+ end
552
+
553
+ class EmbeddingModelMismatch < Gemchat::Error; end
554
+ end
555
+ end