gemchat 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,731 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "optparse"
4
+
5
+ module Gemchat
6
+ class CLI
7
+ # "matched" is absent from `search` and `vsearch` rows and present on
8
+ # `query` rows, where it names the arms that found the chunk. `project` only
9
+ # copies keys a row actually has, so it appears in hybrid JSON and nowhere
10
+ # else without being special-cased.
11
+ RESULT_KEYS = %w[gem_name gem_version source_type title class_name method_name signature source_path source_line url body score matched hosted].freeze
12
+
13
+ def self.start(argv, stdout: $stdout, stderr: $stderr)
14
+ new(stdout:, stderr:).run(argv)
15
+ end
16
+
17
+ def initialize(stdout: $stdout, stderr: $stderr)
18
+ @stdout = stdout
19
+ @stderr = stderr
20
+ end
21
+
22
+ def run(argv)
23
+ command = argv.shift
24
+ case command
25
+ when "init" then cmd_init(argv)
26
+ when "index" then cmd_index(argv)
27
+ when "reindex" then cmd_reindex(argv)
28
+ when "search" then cmd_search(argv)
29
+ when "embed" then cmd_embed(argv)
30
+ when "vsearch" then cmd_vsearch(argv)
31
+ when "query" then cmd_query(argv)
32
+ when "status" then cmd_status(argv)
33
+ when "version" then cmd_version
34
+ when nil, "help", "-h", "--help" then usage
35
+ else
36
+ @stderr.puts("unknown command: #{command}")
37
+ usage
38
+ 1
39
+ end
40
+ rescue OptionParser::ParseError => e
41
+ @stderr.puts(e.message)
42
+ 1
43
+ rescue Gemchat::Error => e
44
+ @stderr.puts("error: #{e.message}")
45
+ 1
46
+ end
47
+
48
+ private
49
+
50
+ def cmd_init(argv)
51
+ opts = {yes: false, help: false, index: true, hosted: false}
52
+ parser = OptionParser.new do |o|
53
+ o.banner = "Usage: gemchat init [options]"
54
+ o.on("-y", "--yes", "Do not prompt") { opts[:yes] = true }
55
+ o.on("--[no-]index", "Index immediately (default: on)") { |v| opts[:index] = v }
56
+ o.on("--hosted", "Record the hosted backend as this project's default") { opts[:hosted] = true }
57
+ o.on("-h", "--help") { opts[:help] = true }
58
+ end
59
+ parser.parse!(argv)
60
+ if opts[:help]
61
+ @stdout.puts(parser)
62
+ return 0
63
+ end
64
+
65
+ result = Init.new(stdout: @stdout, assume_yes: opts[:yes]).run
66
+
67
+ if result.changed?
68
+ result.added.each { |line| @stdout.puts("added to Gemfile: #{line}") }
69
+ else
70
+ @stdout.puts("Gemfile already declares gemchat; nothing to add")
71
+ end
72
+ @stdout.puts("trusted this project: #{Dir.pwd}")
73
+
74
+ if opts[:hosted]
75
+ manifest = Manifest.new
76
+ manifest.hosted = true
77
+ manifest.write
78
+ @stdout.puts("default backend: hosted (gemchat.org); `gemchat query` will use it")
79
+ @stdout.puts(" still needs a key: GEMCHAT_API_KEY, or #{Hosted.credentials_path}")
80
+ end
81
+
82
+ # Trust plus a manifest means the next bundle install is a no-op, so arm
83
+ # the hook before doing any work: a failed index must not leave the user
84
+ # thinking the plugin is live.
85
+ @stdout.puts("")
86
+ @stdout.puts("Run `bundle install` once to arm the plugin hook.")
87
+ @stdout.puts("It then indexes automatically whenever Gemfile.lock changes.")
88
+
89
+ return 0 unless opts[:index]
90
+
91
+ @stdout.puts("")
92
+ cmd_index([])
93
+ end
94
+
95
+ # The remediation `Store::IdentityError` names, so the error is actionable
96
+ # rather than just refusing.
97
+ def cmd_reindex(argv)
98
+ opts = {help: false}
99
+ parser = OptionParser.new do |o|
100
+ o.banner = "Usage: gemchat reindex"
101
+ o.on("-h", "--help") { opts[:help] = true }
102
+ end
103
+ parser.parse!(argv)
104
+ if opts[:help]
105
+ @stdout.puts(parser)
106
+ return 0
107
+ end
108
+
109
+ store = Store.new
110
+ store.close
111
+ FileUtils.rm_f(store.path)
112
+ FileUtils.rm_rf(File.join(Gemchat.root, "ri"))
113
+ @stdout.puts("removed the old index and generated ri cache")
114
+ @stdout.puts("")
115
+ cmd_index([])
116
+ end
117
+
118
+ def cmd_index(argv)
119
+ opts = {include_transitive: false, dry_run: false, only: [], help: false, json: false, all_languages: false}
120
+ parser = OptionParser.new do |o|
121
+ o.banner = "Usage: gemchat index [options]"
122
+ o.on("--all", "Include transitive dependencies") { opts[:include_transitive] = true }
123
+ o.on("--dry-run", "Report what would be indexed, write nothing") { opts[:dry_run] = true }
124
+ o.on("--only=GEMS", Array, "Comma-separated gems to index") { |g| opts[:only] = g }
125
+ o.on("--all-languages", "Also index translated documentation") { opts[:all_languages] = true }
126
+ o.on("--json", "Machine-readable summary on stdout") { opts[:json] = true }
127
+ o.on("-h", "--help") { opts[:help] = true }
128
+ end
129
+ parser.parse!(argv)
130
+ if opts[:help]
131
+ @stdout.puts(parser)
132
+ return 0
133
+ end
134
+
135
+ indexer = Indexer.new(store: Store.new, include_transitive: opts[:include_transitive],
136
+ all_languages: opts[:all_languages])
137
+ specs = indexer.lockfile_specs
138
+ unless opts[:only].empty?
139
+ wanted = Set.new(opts[:only])
140
+ specs = specs.select { |name, _version| wanted.member?(name) }
141
+ end
142
+
143
+ if opts[:dry_run]
144
+ @stdout.puts("lockfile: #{specs.size} gems (#{opts[:include_transitive] ? "all" : "direct"})")
145
+ specs.each { |name, version| @stdout.puts(" would index #{name} #{version}") }
146
+ return 0
147
+ end
148
+
149
+ results = specs.map { |name, version| index_one(indexer, name, version, quiet: opts[:json]) }
150
+ total = results.sum(&:chunks)
151
+ changed = results.count(&:changed?)
152
+ unchanged = results.count { |r| r.status == :unchanged }
153
+ generated = results.count { |r| r.ri_source == "generated" }
154
+ existing = results.count { |r| r.ri_source == "existing" }
155
+ failed = results.reject(&:ok?)
156
+
157
+ # An explicit index run is the thing that grants trust, so running it
158
+ # yourself is how a project opts in to the plugin's auto-indexing.
159
+ Trust.grant!(Dir.pwd, via: "gemchat index")
160
+ write_manifest(specs, opts[:include_transitive])
161
+
162
+ summary = {
163
+ "gems" => results.size,
164
+ "chunks" => total,
165
+ "changed" => changed,
166
+ "unchanged" => unchanged,
167
+ "failed" => failed.size
168
+ }
169
+
170
+ if opts[:json]
171
+ # The last line is the contract the plugin hook parses, so per-gem
172
+ # progress must not be interleaved with it.
173
+ @stdout.puts(JSON.generate(summary))
174
+ else
175
+ @stdout.puts("lockfile: #{results.size} gems (#{opts[:include_transitive] ? "all" : "direct"})")
176
+ @stdout.puts(format("%d gems · %d chunks written (ri: %d existing, %d generated)",
177
+ results.size, total, existing, generated))
178
+ @stdout.puts(format("%d changed, %d unchanged", changed, unchanged)) if unchanged.positive?
179
+ end
180
+
181
+ failed.each { |r| @stderr.puts(" #{r.name}: #{r.error}") }
182
+ failed.empty? ? 0 : 1
183
+ end
184
+
185
+ # Records what the index was built from, so the plugin can answer "is there
186
+ # anything to do?" by reading 64 bytes instead of opening SQLite.
187
+ def write_manifest(specs, _include_transitive)
188
+ digest = Lockfile.digest
189
+ return if digest.nil?
190
+
191
+ Manifest.new(Dir.pwd).record(digest, specs.to_h { |name, version|
192
+ [name, {"version" => version}]
193
+ }).write
194
+ end
195
+
196
+ def index_one(indexer, name, version, quiet: false)
197
+ started = Process.clock_gettime(Process::CLOCK_MONOTONIC)
198
+ result = indexer.index_gem(name, version)
199
+ elapsed = Process.clock_gettime(Process::CLOCK_MONOTONIC) - started
200
+ unless quiet
201
+ label = if result.changed?
202
+ format("%6d chunks", result.chunks)
203
+ else
204
+ format("%6s", result.status)
205
+ end
206
+ @stdout.puts(format(" %-32s %s %-8s %5.2fs", "#{name} #{version}", label, result.ri_source, elapsed))
207
+ end
208
+ result
209
+ end
210
+
211
+ def cmd_search(argv)
212
+ opts = {limit: 10, min_score: 0.0, json: false, help: false}
213
+ parser = OptionParser.new do |o|
214
+ o.banner = "Usage: gemchat search QUERY [options]"
215
+ o.on("-n N", Integer, "Number of results") { |n| opts[:limit] = n }
216
+ o.on("--min-score F", Float, "Minimum score") { |f| opts[:min_score] = f }
217
+ o.on("--json", "JSON output") { opts[:json] = true }
218
+ o.on("-h", "--help") { opts[:help] = true }
219
+ end
220
+ parser.parse!(argv)
221
+ if opts[:help]
222
+ @stdout.puts(parser)
223
+ return 0
224
+ end
225
+
226
+ query = argv.join(" ").strip
227
+ raise Gemchat::Error, "a query is required" if query.empty?
228
+
229
+ results = Store.new.search(query, limit: opts[:limit], min_score: opts[:min_score])
230
+ if opts[:json]
231
+ @stdout.puts(JSON.pretty_generate(results.map { |row| project(row) }))
232
+ else
233
+ results.each_with_index { |row, i| @stdout.puts(format_result(i + 1, row)) }
234
+ @stdout.puts("no results") if results.empty?
235
+ end
236
+ 0
237
+ end
238
+
239
+ def project(row)
240
+ RESULT_KEYS.each_with_object({}) do |key, out|
241
+ out[key] = row[key] if row.key?(key)
242
+ end
243
+ end
244
+
245
+ # The source is shown because it changes how a result should be read: an ri
246
+ # hit is a signature, a readme hit is prose. An agent that cannot tell them
247
+ # apart has to guess whether it is looking at an API or a tutorial.
248
+ def format_result(position, row)
249
+ detail = (row["signature"].to_s.empty? ? row["body"].to_s : row["signature"].to_s)
250
+ summary = detail.lines.first.to_s.strip[0, 100]
251
+ format("%2d. [%-14s %-9s %-7s] %s\n %s%s",
252
+ position, row["gem_name"], row["gem_version"], row["source_type"],
253
+ row["title"], summary, citation(row))
254
+ end
255
+
256
+ # ri says a method exists; the Prism pass says where. Printed when present
257
+ # because it is the one thing a result cannot be used without and cannot be
258
+ # inferred from -- the line to open, and the file to open it in.
259
+ def citation(row)
260
+ path = row["source_path"].to_s
261
+ return "" if path.empty?
262
+
263
+ line = row["source_line"]
264
+ line ? " (#{path}:#{line})" : " (#{path})"
265
+ end
266
+
267
+ # `embed` is the only command that downloads a model, and it is explicit.
268
+ # Search stays BM25-only until it runs, so a fresh install is instantly
269
+ # useful and nothing silently pulls 146MB.
270
+ def cmd_embed(argv)
271
+ opts = {force: false, limit: nil, dry_run: false, help: false, json: false}
272
+ parser = OptionParser.new do |o|
273
+ o.banner = "Usage: gemchat embed [options]"
274
+ o.on("--force", "Discard existing vectors and embed every chunk again") { opts[:force] = true }
275
+ o.on("--limit N", Integer, "Embed at most N chunks") { |n| opts[:limit] = n }
276
+ o.on("--dry-run", "Report what would be embedded") { opts[:dry_run] = true }
277
+ o.on("--json", "Machine-readable summary") { opts[:json] = true }
278
+ o.on("-h", "--help") { opts[:help] = true }
279
+ end
280
+ parser.parse!(argv)
281
+ if opts[:help]
282
+ @stdout.puts(parser)
283
+ return 0
284
+ end
285
+
286
+ name = Models.default
287
+ unless Models.installed?(name)
288
+ # Advisory only. A missing model must not stop the command: it is the
289
+ # one place the download can be triggered from.
290
+ Models.present(name)
291
+ @stdout.puts("downloading #{name} (#{Models.megabytes(name)}MB)…")
292
+ Models.download(name, stdout: @stdout)
293
+ @stdout.puts("")
294
+ end
295
+
296
+ store = Store.new
297
+ identity = Models.identity(name)
298
+ if opts[:force]
299
+ dropped = store.clear_vectors
300
+ @stdout.puts("discarded #{dropped} existing vector(s); embedding every chunk")
301
+ end
302
+
303
+ # Checked before selecting rows. Rows that already hold a vector are
304
+ # excluded from the work list, so with a mismatched model there is nothing
305
+ # to embed and the run would report "nothing to embed" -- true, and useless.
306
+ begin
307
+ store.assert_embedding_model!(Models.identity(name))
308
+ rescue Store::EmbeddingModelMismatch => e
309
+ @stderr.puts(e.message)
310
+ return 1
311
+ end
312
+
313
+ rows = store.unembedded_chunks(limit: opts[:limit])
314
+
315
+ if rows.empty?
316
+ if opts[:json]
317
+ @stdout.puts(JSON.generate({"embedded" => 0, "reason" => "up to date"}))
318
+ else
319
+ @stdout.puts("nothing to embed; every chunk already has a vector")
320
+ end
321
+ return 0
322
+ end
323
+
324
+ texts = rows.map { |r| r["searchable"] }
325
+ if opts[:dry_run]
326
+ @stdout.puts("#{rows.size} chunks would be embedded with #{identity}")
327
+ return 0
328
+ end
329
+
330
+ embedder = Embedder.build(name)
331
+ if embedder.null?
332
+ @stderr.puts("no usable embedder; run `gemchat embed` again")
333
+ return 1
334
+ end
335
+
336
+ # embed returns [index, vector] pairs because a batch can fail even after
337
+ # backoff; only the pairs that came back get written, and the rest stay
338
+ # pending for the next run rather than being misaligned onto the wrong row.
339
+ embedded = embedder.embed(texts)
340
+ written = store.store_embeddings(
341
+ embedded.map { |i, _| rows[i]["id"] },
342
+ embedded.map { |_, vector| vector },
343
+ identity: embedder.identity,
344
+ dimensions: embedder.dimensions
345
+ )
346
+
347
+ if opts[:json]
348
+ @stdout.puts(JSON.generate({"embedded" => written, "model" => embedder.identity}))
349
+ else
350
+ @stdout.puts("embedded #{written} chunks with #{embedder.identity} (#{embedder.dimensions}d)")
351
+ end
352
+ 0
353
+ rescue Store::EmbeddingModelMismatch => e
354
+ @stderr.puts("error: #{e.message}")
355
+ 1
356
+ end
357
+
358
+ def cmd_vsearch(argv)
359
+ opts = {limit: 10, help: false, json: false}
360
+ parser = OptionParser.new do |o|
361
+ o.banner = "Usage: gemchat vsearch QUERY [options]"
362
+ o.on("-n N", Integer, "Number of results") { |n| opts[:limit] = n }
363
+ o.on("--json", "JSON output") { opts[:json] = true }
364
+ o.on("-h", "--help") { opts[:help] = true }
365
+ end
366
+ parser.parse!(argv)
367
+ if opts[:help]
368
+ @stdout.puts(parser)
369
+ return 0
370
+ end
371
+
372
+ query = argv.join(" ").strip
373
+ raise Gemchat::Error, "a query is required" if query.empty?
374
+
375
+ store = Store.new
376
+ unless store.meta("embed_model")
377
+ @stderr.puts("no vectors yet → run `gemchat embed`")
378
+ @stderr.puts("`gemchat search` still works; it uses BM25 and needs no model")
379
+ return 1
380
+ end
381
+
382
+ vector = embed_query(query)
383
+ return 1 if vector.nil?
384
+
385
+ results = store.vector_search(vector, limit: opts[:limit])
386
+ emit(results, opts)
387
+ end
388
+
389
+ # `gemchat query` — hybrid, with the type chosen by a leading token.
390
+ #
391
+ # The three types are not synonyms for "how to spell the query". They pick
392
+ # which arms run, because the two engines fail in opposite directions:
393
+ # measured on the rake index, BM25 returns *no results at all* for a
394
+ # natural-language question, and a vector search is measurably worse than
395
+ # BM25 at an exact identifier. So the default sends a question to both arms
396
+ # and fuses, `lex:` sends identifiers to BM25 alone, and `vec:` sends a
397
+ # paraphrase to the vector arm alone.
398
+ #
399
+ # Degrades to BM25 rather than failing when there are no vectors, because a
400
+ # search command that exits 1 is worse than a search command that answers
401
+ # less well and says so. It never downloads: `embed` is the only command
402
+ # that fetches a model, and a 146MB download in the middle of a query is not
403
+ # something to discover.
404
+ def cmd_query(argv)
405
+ opts = {limit: 10, help: false, json: false, hosted: false, auto_index: false}
406
+ parser = OptionParser.new do |o|
407
+ o.banner = "Usage: gemchat query [lex:|vec:|hyde:] QUESTION [options]"
408
+ o.on("-n N", Integer, "Number of results") { |n| opts[:limit] = n }
409
+ o.on("--json", "JSON output") { opts[:json] = true }
410
+ o.on("--hosted", "Query gemchat.org instead of the local index (explicit; never a fallback)") { opts[:hosted] = true }
411
+ o.on("--auto-index", "With --hosted, ask the server to index gems it does not have") { opts[:auto_index] = true }
412
+ o.on("-h", "--help") { opts[:help] = true }
413
+ end
414
+ parser.parse!(argv)
415
+ if opts[:help]
416
+ @stdout.puts(parser)
417
+ return 0
418
+ end
419
+
420
+ raw = argv.join(" ").strip
421
+ raise Gemchat::Error, "a query is required" if raw.empty?
422
+
423
+ type, question = split_type(raw)
424
+ return hosted_query(question, opts, type: type) if opts[:hosted] || hosted_default?
425
+
426
+ store = Store.new
427
+
428
+ # The type maps straight onto which arms run. `lex:` is BM25 alone, for a
429
+ # caller naming an identifier. The fused default adds the vector arm, and
430
+ # the lexical arm keeps exact-phrase semantics on purpose: see
431
+ # `Store#hybrid_search` for why OR-ing the terms measures worse, not
432
+ # better, and why a bad arm is worse for fusion than no arm.
433
+ plan = {
434
+ lex: {lex: true, lex_mode: :phrase, vector: false},
435
+ vec: {lex: false, lex_mode: :phrase, vector: true},
436
+ hyde: {lex: true, lex_mode: :phrase, vector: true}
437
+ }.fetch(type)
438
+
439
+ has_vectors = !store.meta("embed_model").nil?
440
+ if plan[:vector] && !has_vectors
441
+ @stderr.puts("no vectors yet → run `gemchat embed`")
442
+ @stderr.puts("`gemchat search` still works; it uses BM25 and needs no model")
443
+ # `vec:` asked for semantics, so answering with keyword matches misleads.
444
+ # The fused default is a question, and a question is better served by a
445
+ # weaker answer than by none.
446
+ return 1 if type == :vec
447
+ end
448
+ use_vector = plan[:vector] && has_vectors
449
+
450
+ vector = nil
451
+ if use_vector
452
+ vector = embed_query(question)
453
+ return 1 if vector.nil?
454
+ elsif plan[:vector]
455
+ @stderr.puts("falling back to BM25 only; run `gemchat embed` for hybrid")
456
+ end
457
+
458
+ results = store.hybrid_search(
459
+ question, query_vector: vector,
460
+ lex: plan[:lex], lex_mode: plan[:lex_mode], limit: opts[:limit]
461
+ )
462
+ emit(results, opts)
463
+ end
464
+
465
+ # The hosted path. Explicit only: `--hosted` on the invocation, or a
466
+ # project default set by `gemchat init --hosted`. There is no code path that
467
+ # reaches this without one of those, and a failure here stops rather than
468
+ # quietly answering from the local index -- a user who asked for hosted
469
+ # results and silently got local ones cannot tell the difference.
470
+ #
471
+ # The lockfile's exact versions go with the query. That is the hosted tier's
472
+ # biggest advantage over what §12.3 originally claimed: pins mean the answer
473
+ # is about the version actually installed, and the server discloses any pin
474
+ # it could not apply.
475
+ def hosted_query(question, opts, type: :hyde)
476
+ hosted = Hosted.new
477
+ unless hosted.configured?
478
+ @stderr.puts("hosted backend not configured")
479
+ @stderr.puts(" set GEMCHAT_API_KEY, or write GEMCHAT_API_KEY=... to #{Hosted.credentials_path}")
480
+ @stderr.puts(" keys: gemchat.org/settings")
481
+ @stderr.puts("`gemchat query` without --hosted still searches the local index")
482
+ return 1
483
+ end
484
+
485
+ # Only `hyde`/`vec` have a hosted equivalent. `lex:` is BM25, which §12.2
486
+ # records as local-only, so saying so beats silently returning something else.
487
+ if type == :lex
488
+ @stderr.puts("`lex:` is BM25 and has no hosted equivalent; `gemchat search` covers it locally")
489
+ return 1
490
+ end
491
+
492
+ names, versions = lockfile_pins
493
+ result = hosted.query(question, limit: opts[:limit], gem_names: names, gem_versions: versions,
494
+ auto_index: opts[:auto_index])
495
+ report_hosted(result, opts)
496
+ rescue Hosted::Unauthorized => e
497
+ @stderr.puts("error: #{e.message}")
498
+ 1
499
+ rescue Hosted::Error => e
500
+ @stderr.puts("error: #{e.message}")
501
+ 1
502
+ end
503
+
504
+ # The gem names and exact versions to pin the hosted query to, taken from this
505
+ # project's Gemfile.lock.
506
+ #
507
+ # Direct dependencies only, matching the scope `gemchat index` uses, so the
508
+ # pinned set is the same set the local index holds. Sending all 200
509
+ # transitive specs would be a much larger request to say the same thing.
510
+ #
511
+ # A missing or unparseable lockfile yields empty lists, which the server reads
512
+ # as "search everything" -- degraded, but not wrong, and better than refusing
513
+ # to search because the project is not a Bundler project.
514
+ def lockfile_pins
515
+ path = File.join(Dir.pwd, "Gemfile.lock")
516
+ return [[], []] unless File.file?(path)
517
+
518
+ parsed = ::Bundler::LockfileParser.new(File.read(path))
519
+ specs = parsed.specs.reject { |s| s.name == "ruby-core-stdlib" }
520
+ direct = parsed.dependencies.keys - ["ruby-core-stdlib"]
521
+ specs = specs.select { |s| direct.include?(s.name) }
522
+ [specs.map(&:name), specs.map { |s| s.version.to_s }]
523
+ rescue
524
+ [[], []]
525
+ end
526
+
527
+ def report_hosted(result, opts)
528
+ report_hosted_scope(result.scope) if result.rows.empty?
529
+ @stderr.puts("note: #{result.version_note}") if result.version_note
530
+ @stderr.puts(result.notice) if result.notice
531
+ if result.missing.any? && result.rows.empty?
532
+ @stderr.puts("not indexed on gemchat.org: #{result.missing.join(", ")}")
533
+ @stderr.puts(" re-run with --auto-index to ask the server to index them")
534
+ end
535
+
536
+ rows = result.rows.map { |row| hosted_row(row) }
537
+ emit(rows, opts)
538
+ end
539
+
540
+ # Why there are no results, when the server can say. The two cases look
541
+ # identical on the wire otherwise, and they call for opposite reactions:
542
+ # "nothing matched" means widen the query, "nothing was searchable" means the
543
+ # index is not built and the query was never really asked.
544
+ def report_hosted_scope(scope)
545
+ return if scope.nil?
546
+
547
+ unless scope["searchable"]
548
+ @stderr.puts("nothing in scope is searchable: #{scope["vectorized"]}/#{scope["chunks"]} chunks have vectors")
549
+ @stderr.puts(" this is not an empty result -- the hosted index has no searchable content for this query")
550
+ return
551
+ end
552
+
553
+ @stderr.puts("scope: #{scope["vectorized"]}/#{scope["chunks"]} chunks searchable across #{scope["gems"]} gem(s)")
554
+ end
555
+
556
+ # Hosted rows carry a url or file_path; local rows carry Prism's path:line.
557
+ # Normalised to the local shape so the same renderer and the same `--json`
558
+ # keys work on either backend, with `hosted` marking which one answered.
559
+ def hosted_row(row)
560
+ location = row["url"].to_s.empty? ? row["file_path"].to_s : row["url"].to_s
561
+ {
562
+ "gem_name" => row["gem_name"], "gem_version" => row["gem_version"],
563
+ "source_type" => row["source_type"], "title" => row["title"],
564
+ "signature" => "", "body" => row["body"], "score" => row["score"],
565
+ "source_path" => row["file_path"], "source_line" => nil,
566
+ "url" => row["url"], "matched" => ["hosted"], "hosted" => true
567
+ }.tap { |out| out["url"] = location if location.empty? }
568
+ end
569
+
570
+ QUERY_TYPES = %w[lex vec hyde].freeze
571
+
572
+ # Only a leading `lex:`/`vec:`/`hyde:` is a type. Anything else -- including
573
+ # a query that happens to start `foo: bar` -- is left alone, because
574
+ # misreading a question as a directive is worse than ignoring one.
575
+ def split_type(raw)
576
+ head, rest = raw.split(/\s+/, 2)
577
+ return [:hyde, raw] if head.nil? || rest.nil? || !head.end_with?(":")
578
+
579
+ name = head[0..-2]
580
+ QUERY_TYPES.include?(name) ? [name.to_sym, rest] : [:hyde, raw]
581
+ end
582
+
583
+ # A query is embedded with the model's *query* prefix, not the document one.
584
+ # Returns nil when no model is usable, which the callers treat as "no vectors".
585
+ def embed_query(query)
586
+ return nil if Embedder.build.null?
587
+
588
+ querier = Embedder::Supervised.new(
589
+ model_path: Models.path, identity: Models.identity, prefix: Models.prefix(:query)
590
+ )
591
+ # embed returns [index, vector] pairs, so a single query is .last.
592
+ querier.embed([query]).first&.last
593
+ end
594
+
595
+ def emit(results, opts)
596
+ if opts[:json]
597
+ @stdout.puts(JSON.pretty_generate(results.map { |row| project(row) }))
598
+ else
599
+ results.each_with_index { |row, i| @stdout.puts(format_result(i + 1, row)) }
600
+ @stdout.puts("no results") if results.empty?
601
+ end
602
+ 0
603
+ end
604
+
605
+ def cmd_status(argv)
606
+ opts = {help: false}
607
+ parser = OptionParser.new do |o|
608
+ o.banner = "Usage: gemchat status"
609
+ o.on("-h", "--help") { opts[:help] = true }
610
+ end
611
+ parser.parse!(argv)
612
+ if opts[:help]
613
+ @stdout.puts(parser)
614
+ return 0
615
+ end
616
+
617
+ store = Store.new
618
+ stats = store.stats
619
+ digest = Lockfile.digest
620
+ manifest = Manifest.new
621
+
622
+ @stdout.puts("gemchat: #{Gemchat::VERSION}")
623
+ @stdout.puts("store: #{store.path}")
624
+ @stdout.puts("index: #{stats[:gems]} gems · #{stats[:chunks]} chunks")
625
+ @stdout.puts("project: #{Dir.pwd}")
626
+
627
+ # The three facts that decide whether the plugin will do anything, so
628
+ # "why is auto-index not running" is answerable without a debugger.
629
+ @stdout.puts("lockfile: #{digest ? digest[0, 12] : "none (run `bundle install`)"}")
630
+ @stdout.puts("manifest: #{manifest_state(manifest, digest)}")
631
+ @stdout.puts("trust: #{Trust.trusted? ? "granted" : "not granted → `gemchat init` or `gemchat index`"}")
632
+ @stdout.puts("plugin: #{plugin_state}")
633
+ @stdout.puts("vectors: #{vector_summary(store)}")
634
+ @stdout.puts("hosted: #{hosted_state}")
635
+
636
+ if (drift = store.drift).any?
637
+ @stdout.puts("note: this index was written by a different gemchat " \
638
+ "(#{drift.join(", ")}); reindex if searches look wrong")
639
+ end
640
+ 0
641
+ end
642
+
643
+ def vector_summary(store)
644
+ return "not downloaded (#{Models.megabytes}MB) → gemchat embed" unless Models.installed?
645
+
646
+ # No vectors written yet is a distinct state from "vectors written", and
647
+ # from "downloaded but the index is empty" -- the first asks the user to run
648
+ # embed, the second has nothing to embed.
649
+ return "model downloaded, nothing embedded yet → gemchat embed" unless store.meta("embed_model")
650
+
651
+ stats = store.embedded_stats
652
+ summary = "#{store.meta("embed_model")} · #{stats[:embedded]}/#{stats[:total]} chunks"
653
+ stats[:pending].positive? ? "#{summary} · #{stats[:pending]} pending" : summary
654
+ end
655
+
656
+ # The tier ladder, in one line, naming the action rather than the variable
657
+ # that gates it (§12.4).
658
+ def hosted_state
659
+ host = ENV["GEMCHAT_HOST"].to_s.strip
660
+ return "#{host} (GEMCHAT_HOST)" unless host.empty?
661
+ return "configured via GEMCHAT_API_KEY" unless ENV["GEMCHAT_API_KEY"].to_s.strip.empty?
662
+ return "configured via #{Hosted.credentials_path}" if Hosted.credentials_file
663
+ return "project default: hosted" if hosted_default?
664
+
665
+ "not configured → gemchat.org/settings"
666
+ end
667
+
668
+ # `gemchat init --hosted` records the project's default in the manifest, so a
669
+ # project that wants the hosted tier does not need a flag on every call.
670
+ def hosted_default?
671
+ manifest = Manifest.new
672
+ manifest.exist? && manifest.hosted?
673
+ rescue
674
+ false
675
+ end
676
+
677
+ def manifest_state(manifest, digest)
678
+ return "none → `gemchat index`" unless manifest.exist?
679
+ return "stale → `gemchat index`" unless manifest.up_to_date?(digest)
680
+
681
+ "up to date"
682
+ end
683
+
684
+ # Answers "is the hook armed?", which the Gemfile cannot. A declared plugin
685
+ # only becomes active on the next `bundle install`, and the difference is
686
+ # invisible until you look at Bundler's own index.
687
+ def plugin_state
688
+ return "active" if PluginIndex.armed?
689
+ return "no Gemfile here (run `gemchat init` inside a project)" unless File.file?(Init.new.gemfile_path)
690
+ return "declared in Gemfile → run `bundle install` to arm" if Init.new.declared?(Init::PLUGIN_LINE)
691
+
692
+ "not declared → `gemchat init`"
693
+ end
694
+
695
+ def cmd_version
696
+ @stdout.puts(Gemchat::VERSION)
697
+ 0
698
+ end
699
+
700
+ def usage
701
+ @stdout.puts(<<~TEXT)
702
+ gemchat #{Gemchat::VERSION} — local, offline documentation index for your bundle
703
+
704
+ Usage: gemchat <command> [options]
705
+
706
+ Commands:
707
+ init Set up this project: add gemchat to the Gemfile and trust it
708
+ index Index gems from Gemfile.lock (ri is generated locally)
709
+ reindex Discard the index and generated ri, then index again
710
+ search BM25 full-text search (instant, no models required)
711
+ embed Download the model and embed every chunk (the explicit step)
712
+ vsearch Vector search over the embedded index
713
+ query Hybrid BM25 + vector, fused by reciprocal rank
714
+ status Show index state and what is missing
715
+ version Print the version
716
+
717
+ `query` reads a leading type and routes on it:
718
+
719
+ gemchat query "how do I list every task" both engines, fused (default)
720
+ gemchat query "lex: Rake::Task#enhance" BM25 only
721
+ gemchat query "vec: stop a process on a signal" vectors only
722
+
723
+ `search` works with no model on disk. Everything else needs `gemchat embed`
724
+ to have run first, and only `embed` ever downloads a model.
725
+
726
+ Run `gemchat <command> --help` for command options.
727
+ TEXT
728
+ 0
729
+ end
730
+ end
731
+ end