gemchat 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +7 -0
- data/AGENTS.md +354 -0
- data/CHANGELOG.md +5 -0
- data/CODE_OF_CONDUCT.md +10 -0
- data/LICENSE.txt +21 -0
- data/README.md +144 -0
- data/Rakefile +14 -0
- data/docs/plans/gemchat-local-bundle-index.md +1900 -0
- data/exe/gemchat +7 -0
- data/lib/gemchat/chunker.rb +120 -0
- data/lib/gemchat/cli.rb +731 -0
- data/lib/gemchat/embedder.rb +309 -0
- data/lib/gemchat/env.rb +27 -0
- data/lib/gemchat/errors.rb +10 -0
- data/lib/gemchat/hook.rb +113 -0
- data/lib/gemchat/hosted.rb +195 -0
- data/lib/gemchat/indexer.rb +114 -0
- data/lib/gemchat/init.rb +107 -0
- data/lib/gemchat/lockfile.rb +34 -0
- data/lib/gemchat/manifest.rb +100 -0
- data/lib/gemchat/markdown.rb +103 -0
- data/lib/gemchat/models.rb +241 -0
- data/lib/gemchat/paths.rb +33 -0
- data/lib/gemchat/plugin_index.rb +63 -0
- data/lib/gemchat/prose.rb +211 -0
- data/lib/gemchat/ri.rb +130 -0
- data/lib/gemchat/stopwords.rb +65 -0
- data/lib/gemchat/store.rb +555 -0
- data/lib/gemchat/symbols.rb +159 -0
- data/lib/gemchat/trust.rb +79 -0
- data/lib/gemchat/version.rb +5 -0
- data/lib/gemchat.rb +47 -0
- data/plugins.rb +74 -0
- data/sig/gemchat.rbs +4 -0
- data/skills/gemchat/SKILL.md +158 -0
- metadata +160 -0
|
@@ -0,0 +1,555 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "sqlite3"
|
|
4
|
+
|
|
5
|
+
module Gemchat
|
|
6
|
+
# Local index store: SQLite with FTS5 for BM25. No server, no extension.
|
|
7
|
+
#
|
|
8
|
+
# Vectors are not stored here yet; the schema reserves `embedding` and
|
|
9
|
+
# `embed_model` so §7.1's store identity can refuse to open a store written by
|
|
10
|
+
# a different model rather than silently returning wrong neighbours.
|
|
11
|
+
class Store
|
|
12
|
+
SCHEMA_VERSION = "2"
|
|
13
|
+
BACKEND = "local"
|
|
14
|
+
|
|
15
|
+
# Fields that decide whether a store is readable at all (§7.1).
|
|
16
|
+
#
|
|
17
|
+
# `schema_version` is the hard gate: a newer or older layout cannot be
|
|
18
|
+
# queried correctly, and silently opening it produces wrong answers rather
|
|
19
|
+
# than an error. `gemchat_version` is recorded but tolerated — a point release
|
|
20
|
+
# that only adds a command must not strand a user's index. `embedding_model`
|
|
21
|
+
# and `embedding_dimensions` are placeholders for the embedding phase, where
|
|
22
|
+
# mixing vectors from different models *is* silently wrong.
|
|
23
|
+
IDENTITY_FIELDS = %w[schema_version gemchat_version embedding_model
|
|
24
|
+
embedding_dimensions backend].freeze
|
|
25
|
+
|
|
26
|
+
class IdentityError < Gemchat::Error; end
|
|
27
|
+
|
|
28
|
+
def self.default_path
|
|
29
|
+
File.join(Gemchat.root, "index.sqlite")
|
|
30
|
+
end
|
|
31
|
+
|
|
32
|
+
def initialize(path = self.class.default_path)
|
|
33
|
+
@path = path
|
|
34
|
+
@drift = []
|
|
35
|
+
FileUtils.mkdir_p(File.dirname(path))
|
|
36
|
+
@db = SQLite3::Database.new(path)
|
|
37
|
+
@db.results_as_hash = true
|
|
38
|
+
register_vector_functions
|
|
39
|
+
# Verified before migrating. migrate! rewrites schema_version on every
|
|
40
|
+
# open, so checking afterwards would compare the value against itself and
|
|
41
|
+
# always agree.
|
|
42
|
+
verify_identity!
|
|
43
|
+
migrate!
|
|
44
|
+
record_identity!
|
|
45
|
+
end
|
|
46
|
+
|
|
47
|
+
attr_reader :path
|
|
48
|
+
|
|
49
|
+
# Non-fatal differences worth surfacing in `status`.
|
|
50
|
+
attr_reader :drift
|
|
51
|
+
|
|
52
|
+
def verify_identity!
|
|
53
|
+
return unless table?("meta")
|
|
54
|
+
|
|
55
|
+
stored = meta("schema_version")
|
|
56
|
+
return if stored.nil? || stored == SCHEMA_VERSION
|
|
57
|
+
|
|
58
|
+
raise IdentityError, <<~MSG.strip
|
|
59
|
+
this index was written by a different gemchat (store schema #{stored}, this build expects #{SCHEMA_VERSION}).
|
|
60
|
+
Refusing to read it rather than returning wrong results.
|
|
61
|
+
Rebuild it with: gemchat reindex
|
|
62
|
+
MSG
|
|
63
|
+
end
|
|
64
|
+
|
|
65
|
+
def record_identity!
|
|
66
|
+
# Read before writing. Setting the field first and comparing after would
|
|
67
|
+
# compare the new value against itself and never report drift, which is
|
|
68
|
+
# the one thing this is here to notice.
|
|
69
|
+
previous_version = meta("gemchat_version")
|
|
70
|
+
|
|
71
|
+
set_meta("schema_version", SCHEMA_VERSION)
|
|
72
|
+
set_meta("gemchat_version", Gemchat::VERSION)
|
|
73
|
+
set_meta("backend", BACKEND)
|
|
74
|
+
|
|
75
|
+
@drift << "gemchat_version" if previous_version && previous_version != Gemchat::VERSION
|
|
76
|
+
drift
|
|
77
|
+
end
|
|
78
|
+
|
|
79
|
+
# SQLite has no dot product. This one computes it in C over the float32 blob,
|
|
80
|
+
# which matters: pulling every candidate vector into Ruby to multiply it
|
|
81
|
+
# would be a full scan of the corpus in the interpreter on every query.
|
|
82
|
+
#
|
|
83
|
+
# Dot product equals cosine similarity here because every stored vector is
|
|
84
|
+
# L2-normalised (Rllama::Model#embed normalises by default), so no magnitude
|
|
85
|
+
# term is needed and the result is directly comparable.
|
|
86
|
+
#
|
|
87
|
+
# The result is assigned to the FunctionProxy rather than returned. sqlite3
|
|
88
|
+
# 2.x wraps the block and reads `fp.result`; the block's own return value is
|
|
89
|
+
# discarded. Returning the float instead yields NULL for every row, and
|
|
90
|
+
# because the rank is negated on the way out, that surfaces as a silent 0.0
|
|
91
|
+
# for every result -- correct-looking output, no ordering at all.
|
|
92
|
+
def register_vector_functions
|
|
93
|
+
@db.create_function("dot", 2) do |fp, a, b|
|
|
94
|
+
left = a.to_s.unpack("e*")
|
|
95
|
+
right = b.to_s.unpack("e*")
|
|
96
|
+
fp.result = if left.empty? || left.size != right.size
|
|
97
|
+
nil
|
|
98
|
+
else
|
|
99
|
+
left.zip(right).sum { |x, y| x * y }
|
|
100
|
+
end
|
|
101
|
+
end
|
|
102
|
+
end
|
|
103
|
+
|
|
104
|
+
def table?(name)
|
|
105
|
+
!@db.get_first_value(
|
|
106
|
+
"SELECT 1 FROM sqlite_master WHERE type = 'table' AND name = ?", [name]
|
|
107
|
+
).nil?
|
|
108
|
+
rescue SQLite3::SQLException
|
|
109
|
+
false
|
|
110
|
+
end
|
|
111
|
+
|
|
112
|
+
def migrate!
|
|
113
|
+
@db.execute_batch(<<~SQL)
|
|
114
|
+
CREATE TABLE IF NOT EXISTS meta (
|
|
115
|
+
key TEXT PRIMARY KEY,
|
|
116
|
+
value TEXT
|
|
117
|
+
);
|
|
118
|
+
|
|
119
|
+
CREATE TABLE IF NOT EXISTS gems (
|
|
120
|
+
name TEXT NOT NULL,
|
|
121
|
+
version TEXT NOT NULL,
|
|
122
|
+
source_type TEXT,
|
|
123
|
+
chunk_count INTEGER NOT NULL DEFAULT 0,
|
|
124
|
+
source_hash TEXT,
|
|
125
|
+
indexed_at TEXT,
|
|
126
|
+
PRIMARY KEY (name, version, source_type)
|
|
127
|
+
);
|
|
128
|
+
|
|
129
|
+
CREATE TABLE IF NOT EXISTS chunks (
|
|
130
|
+
id INTEGER PRIMARY KEY,
|
|
131
|
+
gem_name TEXT NOT NULL,
|
|
132
|
+
gem_version TEXT NOT NULL,
|
|
133
|
+
source_type TEXT NOT NULL,
|
|
134
|
+
title TEXT,
|
|
135
|
+
body TEXT,
|
|
136
|
+
class_name TEXT,
|
|
137
|
+
method_name TEXT,
|
|
138
|
+
method_type TEXT,
|
|
139
|
+
signature TEXT,
|
|
140
|
+
source_path TEXT,
|
|
141
|
+
source_line INTEGER,
|
|
142
|
+
content_digest TEXT NOT NULL,
|
|
143
|
+
embedding BLOB,
|
|
144
|
+
embed_model TEXT,
|
|
145
|
+
UNIQUE (gem_name, gem_version, source_type, class_name, method_name, title)
|
|
146
|
+
);
|
|
147
|
+
|
|
148
|
+
CREATE VIRTUAL TABLE IF NOT EXISTS chunks_fts USING fts5(
|
|
149
|
+
title, body, signature, class_name, method_name,
|
|
150
|
+
content='chunks', content_rowid='id', tokenize='porter unicode61'
|
|
151
|
+
);
|
|
152
|
+
SQL
|
|
153
|
+
end
|
|
154
|
+
|
|
155
|
+
def set_meta(key, value)
|
|
156
|
+
@db.execute("INSERT INTO meta (key, value) VALUES (?, ?) ON CONFLICT(key) DO UPDATE SET value = excluded.value", [key, value.to_s])
|
|
157
|
+
end
|
|
158
|
+
|
|
159
|
+
def meta(key)
|
|
160
|
+
row = @db.get_first_row("SELECT value FROM meta WHERE key = ?", [key])
|
|
161
|
+
row&.fetch("value")
|
|
162
|
+
end
|
|
163
|
+
|
|
164
|
+
def indexed?(name, version, source_type = "ri")
|
|
165
|
+
!@db.get_first_value(
|
|
166
|
+
"SELECT 1 FROM gems WHERE name = ? AND version = ? AND source_type = ?",
|
|
167
|
+
[name, version, source_type]
|
|
168
|
+
).nil?
|
|
169
|
+
end
|
|
170
|
+
|
|
171
|
+
def gem_summary(name, version, source_type = "ri")
|
|
172
|
+
@db.get_first_row(
|
|
173
|
+
"SELECT * FROM gems WHERE name = ? AND version = ? AND source_type = ?",
|
|
174
|
+
[name, version, source_type]
|
|
175
|
+
)
|
|
176
|
+
end
|
|
177
|
+
|
|
178
|
+
# What a write actually did. "unchanged" is distinct from "indexed" with
|
|
179
|
+
# zero chunks: the first means the digest matched, the second would mean
|
|
180
|
+
# content was lost, and reporting them the same makes a broken run look
|
|
181
|
+
# like a successful no-op.
|
|
182
|
+
Outcome = Struct.new(:status, :chunks) do
|
|
183
|
+
def changed? = status != :unchanged
|
|
184
|
+
end
|
|
185
|
+
|
|
186
|
+
def replace_gem(name, version, source_type, chunks)
|
|
187
|
+
digest = Digest::SHA256.hexdigest(chunks.map { |c| c.content_digest }.join("|"))
|
|
188
|
+
existing = gem_summary(name, version, source_type)
|
|
189
|
+
return Outcome.new(:unchanged, 0) if existing && existing["source_hash"] == digest
|
|
190
|
+
status = existing ? :updated : :indexed
|
|
191
|
+
|
|
192
|
+
@db.transaction do
|
|
193
|
+
old_ids = @db.execute(
|
|
194
|
+
"SELECT id FROM chunks WHERE gem_name = ? AND gem_version = ? AND source_type = ?",
|
|
195
|
+
[name, version, source_type]
|
|
196
|
+
).map { |r| r["id"] }
|
|
197
|
+
unless old_ids.empty?
|
|
198
|
+
@db.execute("DELETE FROM chunks_fts WHERE rowid IN (#{old_ids.map(&:to_i).join(",")})")
|
|
199
|
+
@db.execute("DELETE FROM chunks WHERE gem_name = ? AND gem_version = ? AND source_type = ?", [name, version, source_type])
|
|
200
|
+
end
|
|
201
|
+
|
|
202
|
+
now = Time.now.utc.iso8601
|
|
203
|
+
chunks.each do |chunk|
|
|
204
|
+
insert_sql = <<~SQL
|
|
205
|
+
INSERT OR IGNORE INTO chunks
|
|
206
|
+
(gem_name, gem_version, source_type, title, body, class_name, method_name,
|
|
207
|
+
method_type, signature, source_path, source_line, content_digest)
|
|
208
|
+
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
|
209
|
+
SQL
|
|
210
|
+
fts_sql = <<~SQL
|
|
211
|
+
INSERT INTO chunks_fts (rowid, title, body, signature, class_name, method_name)
|
|
212
|
+
VALUES (?, ?, ?, ?, ?, ?)
|
|
213
|
+
SQL
|
|
214
|
+
binds = [name, version, source_type, chunk.title, chunk.body,
|
|
215
|
+
chunk.class_name, chunk.method_name, chunk.method_type,
|
|
216
|
+
chunk.signature, chunk.source_path, chunk.source_line,
|
|
217
|
+
chunk.content_digest]
|
|
218
|
+
@db.execute(insert_sql, binds)
|
|
219
|
+
id = @db.last_insert_row_id
|
|
220
|
+
@db.execute(fts_sql, [id, chunk.title, chunk.body, chunk.signature, chunk.class_name, chunk.method_name])
|
|
221
|
+
end
|
|
222
|
+
|
|
223
|
+
@db.execute(<<~SQL, [name, version, source_type, chunks.size, digest, now])
|
|
224
|
+
INSERT INTO gems (name, version, source_type, chunk_count, source_hash, indexed_at)
|
|
225
|
+
VALUES (?, ?, ?, ?, ?, ?)
|
|
226
|
+
ON CONFLICT(name, version, source_type) DO UPDATE SET
|
|
227
|
+
chunk_count = excluded.chunk_count,
|
|
228
|
+
source_hash = excluded.source_hash,
|
|
229
|
+
indexed_at = excluded.indexed_at
|
|
230
|
+
SQL
|
|
231
|
+
end
|
|
232
|
+
|
|
233
|
+
Outcome.new(status, chunks.size)
|
|
234
|
+
end
|
|
235
|
+
|
|
236
|
+
# A gem has one row per source_type, so a plain COUNT(*) would report a gem
|
|
237
|
+
# twice as soon as it also had a README indexed.
|
|
238
|
+
def stats
|
|
239
|
+
{
|
|
240
|
+
gems: @db.get_first_value("SELECT COUNT(DISTINCT name || char(0) || version) FROM gems").to_i,
|
|
241
|
+
chunks: @db.get_first_value("SELECT COUNT(*) FROM chunks").to_i
|
|
242
|
+
}
|
|
243
|
+
end
|
|
244
|
+
|
|
245
|
+
def chunk_counts_by_source
|
|
246
|
+
@db.execute("SELECT source_type, COUNT(*) AS n FROM chunks GROUP BY source_type")
|
|
247
|
+
.to_h { |row| [row["source_type"], row["n"].to_i] }
|
|
248
|
+
end
|
|
249
|
+
|
|
250
|
+
# Explicit column list rather than a splat. `results_as_hash` makes every
|
|
251
|
+
# row carry integer keys too, so `{**row}` would leak 0, 1, 2... into the
|
|
252
|
+
# result and mix symbol and string key styles.
|
|
253
|
+
# `id` is first because fusion needs a stable key: BM25 and vector search
|
|
254
|
+
# return the same chunk as two separate rows, and the identity fields alone
|
|
255
|
+
# are not guaranteed unique. It is not in the CLI's RESULT_KEYS, so it does
|
|
256
|
+
# not leak into `--json` output.
|
|
257
|
+
COLUMNS = %w[id gem_name gem_version source_type title body class_name method_name signature source_path source_line].freeze
|
|
258
|
+
|
|
259
|
+
# Reciprocal rank fusion constant, matching gemchat_app's RRF_K and qmd.
|
|
260
|
+
RRF_K = 60
|
|
261
|
+
|
|
262
|
+
# FTS5 query syntax collides with Ruby naming: an unquoted `Puma::Server`
|
|
263
|
+
# parses as a column filter on `Puma` and fails with "no such column".
|
|
264
|
+
# Quoting each whitespace-separated term makes FTS5 treat it as a phrase and
|
|
265
|
+
# neutralises `:` , `*`, `-`, `"` and friends.
|
|
266
|
+
# The quoting is load-bearing: it is what stops FTS5 reading `Puma::Server` as
|
|
267
|
+
# a column filter, which fails with "no such column" and is a systematic
|
|
268
|
+
# collision with Ruby naming rather than a one-off.
|
|
269
|
+
#
|
|
270
|
+
# But joining the quoted terms with a *space* is FTS5's AND-of-a-phrase, and
|
|
271
|
+
# that was not a decision so much as a side effect of escaping. Measured: on
|
|
272
|
+
# the 236-chunk rake index, a phrase query returns **zero** rows for every
|
|
273
|
+
# natural-language question tried -- "how do I list all the available tasks",
|
|
274
|
+
# "syntax highlighting in the console" -- and a hit only when the query is a
|
|
275
|
+
# literal substring. So `search` was an exact-phrase finder wearing BM25's
|
|
276
|
+
# name, not a keyword search.
|
|
277
|
+
#
|
|
278
|
+
# :phrase the quoted run. Exact substrings, high precision, near-zero recall.
|
|
279
|
+
# The default for `search`, where a caller asking for
|
|
280
|
+
# "Rake::Task#enhance" wants exactly that. Stopwords are never
|
|
281
|
+
# dropped here: removing a word from a literal phrase changes what
|
|
282
|
+
# that phrase matches, which is a different query, not a better one.
|
|
283
|
+
# :any the same terms OR'd, with stopwords dropped. What a keyword search
|
|
284
|
+
# means, and what the fusion's lexical arm needs. See §11.3 for what
|
|
285
|
+
# it measures and §11.4 for the stopword pass and the eval gate.
|
|
286
|
+
def self.fts_query(raw, mode: :phrase)
|
|
287
|
+
terms = raw.to_s.split(/\s+/).reject(&:empty?)
|
|
288
|
+
return nil if terms.empty?
|
|
289
|
+
|
|
290
|
+
if mode == :any
|
|
291
|
+
kept = terms.uniq.reject { |term| Stopwords.only_stopwords?(term) }
|
|
292
|
+
# Every term was a stopword. Fall back to the full set rather than
|
|
293
|
+
# returning nil: the caller asked a real question, and an empty query
|
|
294
|
+
# would report "no results" as though the corpus were empty.
|
|
295
|
+
kept = terms.uniq if kept.empty?
|
|
296
|
+
kept.map { |term| %("#{term.gsub('"', '""')}") }.join(" OR ")
|
|
297
|
+
else
|
|
298
|
+
terms.uniq.map { |term| %("#{term.gsub('"', '""')}") }.join(" ")
|
|
299
|
+
end
|
|
300
|
+
end
|
|
301
|
+
|
|
302
|
+
# BM25's rank() is lower-is-better, so results are exposed as a positive
|
|
303
|
+
# `score` where higher is better. Every key is a String, including "score".
|
|
304
|
+
def search(query, limit: 10, min_score: 0.0, gem_names: nil, mode: :phrase)
|
|
305
|
+
match = self.class.fts_query(query, mode: mode)
|
|
306
|
+
return [] if match.nil?
|
|
307
|
+
|
|
308
|
+
binds = [match]
|
|
309
|
+
scope = ""
|
|
310
|
+
if gem_names && !gem_names.empty?
|
|
311
|
+
scope = "AND c.gem_name IN (#{(["?"] * gem_names.size).join(", ")})"
|
|
312
|
+
binds.concat(gem_names)
|
|
313
|
+
end
|
|
314
|
+
binds << limit
|
|
315
|
+
|
|
316
|
+
sql = <<~SQL
|
|
317
|
+
SELECT #{COLUMNS.map { |c| "c.#{c}" }.join(", ")},
|
|
318
|
+
bm25(chunks_fts) AS rank
|
|
319
|
+
FROM chunks_fts
|
|
320
|
+
JOIN chunks c ON c.id = chunks_fts.rowid
|
|
321
|
+
WHERE chunks_fts MATCH ? #{scope}
|
|
322
|
+
ORDER BY rank
|
|
323
|
+
LIMIT ?
|
|
324
|
+
SQL
|
|
325
|
+
|
|
326
|
+
@db.execute(sql, binds).map { |row|
|
|
327
|
+
COLUMNS.to_h { |c| [c, row[c]] }.merge("score" => -row["rank"].to_f)
|
|
328
|
+
}.select { |row| row["score"] >= min_score }
|
|
329
|
+
end
|
|
330
|
+
|
|
331
|
+
# --- vectors ------------------------------------------------------------
|
|
332
|
+
|
|
333
|
+
# Refuse to write vectors from a model this store has not seen. Two models in
|
|
334
|
+
# one table cannot be compared, and the failure would be silent: cosine
|
|
335
|
+
# between unrelated spaces still returns a number, just a meaningless one.
|
|
336
|
+
# Drops every vector and the metadata that identifies them. Needed because
|
|
337
|
+
# `unembedded_chunks` only returns rows where the vector IS NULL, so a model
|
|
338
|
+
# change cannot be reconciled by re-running `embed` -- without this the
|
|
339
|
+
# mismatch error would tell the user to run a command that does nothing.
|
|
340
|
+
def clear_vectors
|
|
341
|
+
@db.execute("UPDATE chunks SET embedding = NULL WHERE embedding IS NOT NULL")
|
|
342
|
+
dropped = @db.changes
|
|
343
|
+
@db.execute("DELETE FROM meta WHERE key IN ('embed_model', 'embed_dimensions', 'embed_at')")
|
|
344
|
+
dropped
|
|
345
|
+
end
|
|
346
|
+
|
|
347
|
+
def assert_embedding_model!(identity)
|
|
348
|
+
stored = meta("embed_model")
|
|
349
|
+
return true if stored.nil? || stored == identity
|
|
350
|
+
|
|
351
|
+
raise EmbeddingModelMismatch, <<~MSG.strip
|
|
352
|
+
this index already holds vectors from #{stored}, and the current model is #{identity}.
|
|
353
|
+
Vectors from different models cannot be compared, so the store will not mix them.
|
|
354
|
+
Discard the vectors with: gemchat embed --force
|
|
355
|
+
MSG
|
|
356
|
+
end
|
|
357
|
+
|
|
358
|
+
# float32 little-endian: 768 dimensions is 3072 bytes a chunk, so ~43MB for
|
|
359
|
+
# the 14k-chunk corpus the plan measured. Small enough to keep in the row.
|
|
360
|
+
def pack(vector)
|
|
361
|
+
vector.map { |v| v.to_f }.pack("e*")
|
|
362
|
+
end
|
|
363
|
+
|
|
364
|
+
def unpack(blob)
|
|
365
|
+
return [] if blob.nil? || blob.empty?
|
|
366
|
+
|
|
367
|
+
blob.unpack("e*")
|
|
368
|
+
end
|
|
369
|
+
|
|
370
|
+
def store_embeddings(rows, vectors, identity:, dimensions:)
|
|
371
|
+
assert_embedding_model!(identity)
|
|
372
|
+
return 0 if rows.empty?
|
|
373
|
+
|
|
374
|
+
unless vectors.size == rows.size
|
|
375
|
+
raise Error, "embedding count mismatch: #{vectors.size} vectors for #{rows.size} chunks"
|
|
376
|
+
end
|
|
377
|
+
|
|
378
|
+
bad = vectors.find { |v| v.size != dimensions }
|
|
379
|
+
if bad
|
|
380
|
+
raise Error, "expected #{dimensions}-dimension vectors, got #{bad.size}"
|
|
381
|
+
end
|
|
382
|
+
|
|
383
|
+
now = Time.now.utc.iso8601
|
|
384
|
+
@db.transaction do
|
|
385
|
+
rows.each_with_index do |id, i|
|
|
386
|
+
@db.execute("UPDATE chunks SET embedding = ?, embed_model = ? WHERE id = ?",
|
|
387
|
+
[pack(vectors[i]), identity, id])
|
|
388
|
+
end
|
|
389
|
+
end
|
|
390
|
+
|
|
391
|
+
set_meta("embed_model", identity)
|
|
392
|
+
set_meta("embedding_dimensions", dimensions)
|
|
393
|
+
set_meta("embedded_at", now)
|
|
394
|
+
vectors.size
|
|
395
|
+
end
|
|
396
|
+
|
|
397
|
+
# The chunks still needing a vector, with the text to embed. Ordered so a
|
|
398
|
+
# resumed run picks up where it stopped.
|
|
399
|
+
#
|
|
400
|
+
# The text is rebuilt through Chunker::Chunk#searchable_text rather than
|
|
401
|
+
# assembled in SQL. A query embedding and a document embedding must be
|
|
402
|
+
# produced by *identical* text, and a second implementation of that format in
|
|
403
|
+
# SQL would drift from the Ruby one the chunks were written with -- silently,
|
|
404
|
+
# because both still produce a valid-looking vector.
|
|
405
|
+
def unembedded_chunks(limit: nil)
|
|
406
|
+
sql = <<~SQL
|
|
407
|
+
SELECT id, gem_name, gem_version, source_type, title, body,
|
|
408
|
+
class_name, method_name, method_type, signature
|
|
409
|
+
FROM chunks
|
|
410
|
+
WHERE embedding IS NULL
|
|
411
|
+
ORDER BY id
|
|
412
|
+
SQL
|
|
413
|
+
sql += " LIMIT #{limit.to_i}" if limit
|
|
414
|
+
|
|
415
|
+
@db.execute(sql).map { |row|
|
|
416
|
+
chunk = Chunker::Chunk.new(
|
|
417
|
+
gem_name: row["gem_name"], gem_version: row["gem_version"],
|
|
418
|
+
source_type: row["source_type"], title: row["title"], body: row["body"],
|
|
419
|
+
class_name: row["class_name"], method_name: row["method_name"],
|
|
420
|
+
method_type: row["method_type"], signature: row["signature"]
|
|
421
|
+
)
|
|
422
|
+
{"id" => row["id"], "searchable" => chunk.searchable_text}
|
|
423
|
+
}
|
|
424
|
+
end
|
|
425
|
+
|
|
426
|
+
def embedded_stats
|
|
427
|
+
total = @db.get_first_value("SELECT COUNT(*) FROM chunks").to_i
|
|
428
|
+
embedded = @db.get_first_value("SELECT COUNT(*) FROM chunks WHERE embedding IS NOT NULL").to_i
|
|
429
|
+
{total:, embedded:, pending: total - embedded}
|
|
430
|
+
end
|
|
431
|
+
|
|
432
|
+
# Unit vectors, so a dot product is cosine similarity and the ordering falls
|
|
433
|
+
# out of a single pass. Scoped to the same model as the query was embedded
|
|
434
|
+
# with, so a mixed store cannot return a comparison across spaces.
|
|
435
|
+
def vector_search(query_vector, limit: 10, min_score: 0.0, gem_names: nil)
|
|
436
|
+
return [] if query_vector.nil? || query_vector.empty?
|
|
437
|
+
|
|
438
|
+
identity = meta("embed_model")
|
|
439
|
+
return [] if identity.nil?
|
|
440
|
+
|
|
441
|
+
# A mis-sized query makes dot() return NULL for every row, which sorts as
|
|
442
|
+
# 0.0 and would return an arbitrary slice of the corpus as if it were a
|
|
443
|
+
# ranked answer. Refusing up front turns a silent wrong answer into none.
|
|
444
|
+
stored_dimensions = meta("embedding_dimensions")&.to_i
|
|
445
|
+
if stored_dimensions && stored_dimensions != query_vector.size
|
|
446
|
+
return []
|
|
447
|
+
end
|
|
448
|
+
|
|
449
|
+
binds = [pack(query_vector), identity]
|
|
450
|
+
scope = ""
|
|
451
|
+
if gem_names && !gem_names.empty?
|
|
452
|
+
scope = "AND c.gem_name IN (#{(["?"] * gem_names.size).join(", ")})"
|
|
453
|
+
binds.concat(gem_names)
|
|
454
|
+
end
|
|
455
|
+
binds << limit
|
|
456
|
+
|
|
457
|
+
sql = <<~SQL
|
|
458
|
+
SELECT #{COLUMNS.map { |c| "c.#{c}" }.join(", ")},
|
|
459
|
+
-dot(embedding, ?) AS rank
|
|
460
|
+
FROM chunks c
|
|
461
|
+
WHERE c.embedding IS NOT NULL
|
|
462
|
+
AND c.embed_model = ? #{scope}
|
|
463
|
+
ORDER BY rank
|
|
464
|
+
LIMIT ?
|
|
465
|
+
SQL
|
|
466
|
+
|
|
467
|
+
@db.execute(sql, binds).map { |row|
|
|
468
|
+
COLUMNS.to_h { |c| [c, row[c]] }.merge("score" => -row["rank"].to_f)
|
|
469
|
+
}.select { |row| row["score"] >= min_score }
|
|
470
|
+
end
|
|
471
|
+
|
|
472
|
+
# --- hybrid -------------------------------------------------------------
|
|
473
|
+
|
|
474
|
+
# Runs whichever arms were asked for and fuses them by reciprocal rank.
|
|
475
|
+
#
|
|
476
|
+
# The UDF constraint from above does not bite here. It stopped `vector_search`
|
|
477
|
+
# from being narrowed by FTS inside a single statement, but each arm here
|
|
478
|
+
# returns a small ranked list on its own, so the fusion is a hash fold over a
|
|
479
|
+
# few hundred rows in Ruby. That is a different operation from the one that
|
|
480
|
+
# was infeasible, and worth distinguishing before assuming otherwise.
|
|
481
|
+
#
|
|
482
|
+
# RRF is used precisely because the two scores are not comparable: one is a
|
|
483
|
+
# negated BM25 rank on an unbounded scale, the other a cosine in [-1, 1].
|
|
484
|
+
# Fusing on *rank* sidesteps normalisation entirely, which is the reason to
|
|
485
|
+
# reach for it rather than for a weighted sum of the raw scores.
|
|
486
|
+
# `lex_mode` is a parameter rather than a constant because the obvious
|
|
487
|
+
# improvement to :phrase is wrong, and that is worth being able to measure
|
|
488
|
+
# rather than re-derive:
|
|
489
|
+
#
|
|
490
|
+
# :phrase the quoted run. Measured on 22,320 chunks from 34 gems, it
|
|
491
|
+
# returns zero rows for every natural-language question tried, and
|
|
492
|
+
# one exact hit for an identifier. High precision, no recall.
|
|
493
|
+
# :any the same terms OR'd. This is *worse*, not better: on five
|
|
494
|
+
# known-answer queries -- queries whose answer definitely exists in
|
|
495
|
+
# the corpus -- it located the right chunk in 1 of 5. Short ri
|
|
496
|
+
# signatures containing one query word outrank the long prose chunk
|
|
497
|
+
# that actually answers the question, because a rare term in a short
|
|
498
|
+
# document scores enormously under BM25.
|
|
499
|
+
#
|
|
500
|
+
# So :any is not the default. Feeding it to a fusion is actively harmful,
|
|
501
|
+
# because RRF trusts rank position: a confidently wrong arm does not get
|
|
502
|
+
# ignored, it gets promoted. A real keyword arm needs term selection (drop
|
|
503
|
+
# the common words, require a rare term) before it belongs anywhere near
|
|
504
|
+
# :any. See §11.3 of the plan.
|
|
505
|
+
def hybrid_search(query, query_vector: nil, lex: true, lex_mode: :phrase,
|
|
506
|
+
limit: 10, pool: 50, k: RRF_K, gem_names: nil)
|
|
507
|
+
arms = []
|
|
508
|
+
|
|
509
|
+
if lex
|
|
510
|
+
found = search(query, limit: pool, gem_names: gem_names, mode: lex_mode)
|
|
511
|
+
arms << [:bm25, found] unless found.empty?
|
|
512
|
+
end
|
|
513
|
+
|
|
514
|
+
if query_vector
|
|
515
|
+
vec = vector_search(query_vector, limit: pool, gem_names: gem_names)
|
|
516
|
+
arms << [:vector, vec] unless vec.empty?
|
|
517
|
+
end
|
|
518
|
+
|
|
519
|
+
return [] if arms.empty?
|
|
520
|
+
|
|
521
|
+
rrf_fuse(arms, limit: limit, k: k)
|
|
522
|
+
end
|
|
523
|
+
|
|
524
|
+
# `arms` is [[name, ranked_rows], ...]. Each row carries "matched", naming
|
|
525
|
+
# the arms that found it, because two engines agreeing is signal in its own
|
|
526
|
+
# right and the fold would otherwise throw it away.
|
|
527
|
+
def rrf_fuse(arms, limit:, k: RRF_K)
|
|
528
|
+
scores = Hash.new(0.0)
|
|
529
|
+
matched = Hash.new { |h, key| h[key] = [] }
|
|
530
|
+
rows = {}
|
|
531
|
+
|
|
532
|
+
arms.each do |name, list|
|
|
533
|
+
list.each_with_index do |row, i|
|
|
534
|
+
id = row["id"]
|
|
535
|
+
scores[id] += 1.0 / (k + i + 1)
|
|
536
|
+
matched[id] << name
|
|
537
|
+
rows[id] ||= row
|
|
538
|
+
end
|
|
539
|
+
end
|
|
540
|
+
|
|
541
|
+
# Ties break on id so the order is deterministic across runs. Without this,
|
|
542
|
+
# two equally-scored chunks swap places between invocations, which makes a
|
|
543
|
+
# regression look like a change.
|
|
544
|
+
rows.keys.sort_by { |id| [-scores[id], id] }.first(limit).map { |id|
|
|
545
|
+
rows[id].merge("score" => scores[id], "matched" => matched[id])
|
|
546
|
+
}
|
|
547
|
+
end
|
|
548
|
+
|
|
549
|
+
def close
|
|
550
|
+
@db.close
|
|
551
|
+
end
|
|
552
|
+
|
|
553
|
+
class EmbeddingModelMismatch < Gemchat::Error; end
|
|
554
|
+
end
|
|
555
|
+
end
|