gemchat 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +7 -0
- data/AGENTS.md +354 -0
- data/CHANGELOG.md +5 -0
- data/CODE_OF_CONDUCT.md +10 -0
- data/LICENSE.txt +21 -0
- data/README.md +144 -0
- data/Rakefile +14 -0
- data/docs/plans/gemchat-local-bundle-index.md +1900 -0
- data/exe/gemchat +7 -0
- data/lib/gemchat/chunker.rb +120 -0
- data/lib/gemchat/cli.rb +731 -0
- data/lib/gemchat/embedder.rb +309 -0
- data/lib/gemchat/env.rb +27 -0
- data/lib/gemchat/errors.rb +10 -0
- data/lib/gemchat/hook.rb +113 -0
- data/lib/gemchat/hosted.rb +195 -0
- data/lib/gemchat/indexer.rb +114 -0
- data/lib/gemchat/init.rb +107 -0
- data/lib/gemchat/lockfile.rb +34 -0
- data/lib/gemchat/manifest.rb +100 -0
- data/lib/gemchat/markdown.rb +103 -0
- data/lib/gemchat/models.rb +241 -0
- data/lib/gemchat/paths.rb +33 -0
- data/lib/gemchat/plugin_index.rb +63 -0
- data/lib/gemchat/prose.rb +211 -0
- data/lib/gemchat/ri.rb +130 -0
- data/lib/gemchat/stopwords.rb +65 -0
- data/lib/gemchat/store.rb +555 -0
- data/lib/gemchat/symbols.rb +159 -0
- data/lib/gemchat/trust.rb +79 -0
- data/lib/gemchat/version.rb +5 -0
- data/lib/gemchat.rb +47 -0
- data/plugins.rb +74 -0
- data/sig/gemchat.rbs +4 -0
- data/skills/gemchat/SKILL.md +158 -0
- metadata +160 -0
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Gemchat
|
|
4
|
+
# Where things live on disk.
|
|
5
|
+
#
|
|
6
|
+
# Its own file so that anything loaded on its own — notably the Bundler plugin
|
|
7
|
+
# shim, which runs in a fresh child during `bundle install` — can resolve a path
|
|
8
|
+
# without requiring the whole gem, and without a circular require back through
|
|
9
|
+
# lib/gemchat.rb.
|
|
10
|
+
module Paths
|
|
11
|
+
module_function
|
|
12
|
+
|
|
13
|
+
# Index location: the ri cache and index.sqlite.
|
|
14
|
+
#
|
|
15
|
+
# GEMCHAT_HOME is taken verbatim. An override that silently grew a suffix
|
|
16
|
+
# would make it impossible to predict where the data went.
|
|
17
|
+
#
|
|
18
|
+
# Deliberately not memoised. Memoising a value derived from the environment
|
|
19
|
+
# means the first reader wins for the life of the process, so setting
|
|
20
|
+
# GEMCHAT_HOME later does nothing — which once made the test suite write to
|
|
21
|
+
# the developer's real ~/.gemchat. The join is cheap enough to repeat.
|
|
22
|
+
def root
|
|
23
|
+
File.expand_path(ENV["GEMCHAT_HOME"] || File.join(Dir.home, ".gemchat"))
|
|
24
|
+
end
|
|
25
|
+
|
|
26
|
+
# Settings and trust records, kept separate from the index so wiping the
|
|
27
|
+
# cache does not also discard the decisions about which projects to trust.
|
|
28
|
+
def config_home
|
|
29
|
+
ENV["GEMCHAT_CONFIG_HOME"] ||
|
|
30
|
+
File.join(ENV["XDG_CONFIG_HOME"] || File.join(Dir.home, ".config"), "gemchat")
|
|
31
|
+
end
|
|
32
|
+
end
|
|
33
|
+
end
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "yaml"
|
|
4
|
+
|
|
5
|
+
module Gemchat
|
|
6
|
+
# Reads Bundler's own record of which plugins are installed and which hooks
|
|
7
|
+
# they registered.
|
|
8
|
+
#
|
|
9
|
+
# This exists because "is the hook actually armed?" cannot be answered by
|
|
10
|
+
# looking at the Gemfile. A Gemfile line only says the plugin was *requested*;
|
|
11
|
+
# Bundler activates it on the next install, and only writes this index when it
|
|
12
|
+
# does. The difference is the whole point — §15.1a lists four ways a plugin
|
|
13
|
+
# looks installed and does nothing, and `hooks:` being empty here is the
|
|
14
|
+
# signature of the worst one: a plugin that ran once, during the install that
|
|
15
|
+
# installed it, and never again.
|
|
16
|
+
module PluginIndex
|
|
17
|
+
FILENAME = "index"
|
|
18
|
+
|
|
19
|
+
module_function
|
|
20
|
+
|
|
21
|
+
# Honours BUNDLE_APP_CONFIG the way Bundler does, so this reads the same
|
|
22
|
+
# index the hook does.
|
|
23
|
+
def dir
|
|
24
|
+
config = ENV["BUNDLE_APP_CONFIG"]
|
|
25
|
+
config ||= File.join(Dir.pwd, ".bundle")
|
|
26
|
+
File.join(config, "plugin")
|
|
27
|
+
end
|
|
28
|
+
|
|
29
|
+
def path
|
|
30
|
+
File.join(dir, FILENAME)
|
|
31
|
+
end
|
|
32
|
+
|
|
33
|
+
def exist?
|
|
34
|
+
File.file?(path)
|
|
35
|
+
end
|
|
36
|
+
|
|
37
|
+
def data
|
|
38
|
+
return nil unless exist?
|
|
39
|
+
|
|
40
|
+
parsed = YAML.safe_load_file(path, permitted_classes: [Symbol], aliases: true)
|
|
41
|
+
parsed.is_a?(Hash) ? parsed : nil
|
|
42
|
+
rescue
|
|
43
|
+
nil
|
|
44
|
+
end
|
|
45
|
+
|
|
46
|
+
def installed_plugins
|
|
47
|
+
data&.dig("plugin_paths")&.keys || []
|
|
48
|
+
end
|
|
49
|
+
|
|
50
|
+
def registered_hooks
|
|
51
|
+
data&.dig("hooks") || {}
|
|
52
|
+
end
|
|
53
|
+
|
|
54
|
+
def hook?(name, event = "after-install-all")
|
|
55
|
+
Array(registered_hooks[event]).include?(name)
|
|
56
|
+
end
|
|
57
|
+
|
|
58
|
+
# Whether the hook will actually run, as opposed to being merely declared.
|
|
59
|
+
def armed?(name = "gemchat", event: "after-install-all")
|
|
60
|
+
hook?(name, event)
|
|
61
|
+
end
|
|
62
|
+
end
|
|
63
|
+
end
|
|
@@ -0,0 +1,211 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require_relative "markdown"
|
|
4
|
+
|
|
5
|
+
module Gemchat
|
|
6
|
+
# The prose half of a gem: its README, its guides, and the reference material
|
|
7
|
+
# it ships under `doc/`.
|
|
8
|
+
#
|
|
9
|
+
# ri answers "what does this method take and return". Prose answers "how is
|
|
10
|
+
# this library actually used", which is the question an agent most often has
|
|
11
|
+
# and the one ri is worst at. Each source is stored under its own
|
|
12
|
+
# `source_type`, so a gem's README and its reference manual never overwrite one
|
|
13
|
+
# another and search output can say what kind of document a hit came from.
|
|
14
|
+
#
|
|
15
|
+
# The `readme` / `guides` / `doc` split is not incidental: `gemchat_app` already
|
|
16
|
+
# uses exactly this vocabulary, gives `guides` a ranking prior of 1.15, and
|
|
17
|
+
# groups `readme`+`guides` only at query time to enforce prose diversity. See
|
|
18
|
+
# §20.1 of the plan.
|
|
19
|
+
class Prose
|
|
20
|
+
FILENAME = "README"
|
|
21
|
+
PROSE_EXTENSIONS = %w[.md .markdown .mkd .rdoc .txt .textile].freeze
|
|
22
|
+
GUIDE_DIRS = %w[guides guide].freeze
|
|
23
|
+
DOC_DIRS = %w[doc docs].freeze
|
|
24
|
+
|
|
25
|
+
# Generated output. A gem installed with documentation has ri under
|
|
26
|
+
# `doc/<name>-<version>/ri`, and older gemspecs committed rdoc's HTML tree.
|
|
27
|
+
# Both duplicate what the ri half already covers, so both are dropped — but
|
|
28
|
+
# only from `doc/`, and only by shape. Hand-written files are kept.
|
|
29
|
+
GENERATED_DIRS = %w[ri].freeze
|
|
30
|
+
GENERATED_EXTENSIONS = %w[.html .htm .js .css].freeze
|
|
31
|
+
|
|
32
|
+
# Languages seen in bundled documentation, keyed by the marker that appears
|
|
33
|
+
# in a filename or directory name. Skipped by default: they cost embedding
|
|
34
|
+
# time and can never match an English query, while polluting BM25 with tokens
|
|
35
|
+
# no English query contains. `en` is deliberately absent.
|
|
36
|
+
FOREIGN_MARKERS = %w[
|
|
37
|
+
ja zh ko fr de es pt it ru pl tr nl sv da no fi cs hu el he ar hi th vi
|
|
38
|
+
id uk ro bg sr hr sk sl et lv lt ca gl eu af sw fa ur bn ta ms tl
|
|
39
|
+
].freeze
|
|
40
|
+
|
|
41
|
+
Document = Struct.new(:source_type, :label, :path, :text)
|
|
42
|
+
|
|
43
|
+
class << self
|
|
44
|
+
def chunks(gem_name, gem_version, gem_dir, all_languages: false)
|
|
45
|
+
documents(gem_dir, all_languages: all_languages)
|
|
46
|
+
.flat_map { |doc| chunks_for(gem_name, gem_version, doc) }
|
|
47
|
+
end
|
|
48
|
+
|
|
49
|
+
def documents(gem_dir, all_languages: false)
|
|
50
|
+
paths(gem_dir, all_languages: all_languages).map { |path|
|
|
51
|
+
Document.new(source_type_for(gem_dir, path), label_for(gem_dir, path),
|
|
52
|
+
path, read(path))
|
|
53
|
+
}
|
|
54
|
+
end
|
|
55
|
+
|
|
56
|
+
def paths(gem_dir, all_languages: false)
|
|
57
|
+
return [] unless File.directory?(gem_dir)
|
|
58
|
+
|
|
59
|
+
all = [readme_in(gem_dir)] +
|
|
60
|
+
within(gem_dir, GUIDE_DIRS) +
|
|
61
|
+
within(gem_dir, DOC_DIRS)
|
|
62
|
+
|
|
63
|
+
all.compact.uniq.select { |path| indexable?(path, all_languages) }
|
|
64
|
+
end
|
|
65
|
+
|
|
66
|
+
def source_type_for(gem_dir, path)
|
|
67
|
+
relative = path.sub("#{gem_dir}/", "")
|
|
68
|
+
return "readme" unless relative.include?("/")
|
|
69
|
+
|
|
70
|
+
first = relative.split("/").first
|
|
71
|
+
GUIDE_DIRS.include?(first) ? "guides" : "doc"
|
|
72
|
+
end
|
|
73
|
+
|
|
74
|
+
# The directory the document lives in, relative to the gem root. Keeps
|
|
75
|
+
# `doc/rexml/parsers.md` and `doc/csv/io.rdoc` distinguishable in a result
|
|
76
|
+
# title, which "doc" alone would not.
|
|
77
|
+
def label_for(gem_dir, path)
|
|
78
|
+
relative = path.sub("#{gem_dir}/", "")
|
|
79
|
+
return "README" unless relative.include?("/")
|
|
80
|
+
|
|
81
|
+
File.dirname(relative)
|
|
82
|
+
end
|
|
83
|
+
|
|
84
|
+
private
|
|
85
|
+
|
|
86
|
+
def within(gem_dir, dirs)
|
|
87
|
+
dirs.flat_map { |d| Dir.glob(File.join(gem_dir, d, "**", "*")) }
|
|
88
|
+
end
|
|
89
|
+
|
|
90
|
+
def readme_in(gem_dir)
|
|
91
|
+
Prose::PROSE_EXTENSIONS.each do |ext|
|
|
92
|
+
candidate = File.join(gem_dir, "#{FILENAME}#{ext}")
|
|
93
|
+
return candidate if File.file?(candidate)
|
|
94
|
+
end
|
|
95
|
+
bare = File.join(gem_dir, FILENAME)
|
|
96
|
+
bare if File.file?(bare)
|
|
97
|
+
end
|
|
98
|
+
|
|
99
|
+
def indexable?(path, all_languages)
|
|
100
|
+
return false unless File.file?(path)
|
|
101
|
+
return false if foreign?(path) && !all_languages
|
|
102
|
+
# A README is taken whatever its extension; everything under a guide or
|
|
103
|
+
# doc directory has to look like prose, or the tree fills up with
|
|
104
|
+
# example sources and translation catalogues.
|
|
105
|
+
return true if File.basename(path).start_with?(FILENAME.downcase)
|
|
106
|
+
return false unless PROSE_EXTENSIONS.include?(extension(path))
|
|
107
|
+
|
|
108
|
+
!generated?(path)
|
|
109
|
+
end
|
|
110
|
+
|
|
111
|
+
def extension(path)
|
|
112
|
+
File.extname(path).downcase
|
|
113
|
+
end
|
|
114
|
+
|
|
115
|
+
def generated?(path)
|
|
116
|
+
parts = path.split("/")
|
|
117
|
+
return true if parts.any? { |p| GENERATED_DIRS.include?(p) }
|
|
118
|
+
|
|
119
|
+
GENERATED_EXTENSIONS.include?(extension(path))
|
|
120
|
+
end
|
|
121
|
+
|
|
122
|
+
# Matches a language marker only when it stands alone as a path segment or
|
|
123
|
+
# as a dot- or underscore-delimited token in the basename, so a directory
|
|
124
|
+
# called `text` or `example` is not mistaken for one.
|
|
125
|
+
def foreign?(path)
|
|
126
|
+
parts = path.split("/")
|
|
127
|
+
base = File.basename(parts.pop, ".*").downcase
|
|
128
|
+
tokens = base.split(/[._-]/).map { |p| p.sub(/\.(rd|md|txt|textile)$/, "") }
|
|
129
|
+
return true if (parts + tokens).any? { |p| FOREIGN_MARKERS.include?(p.downcase) }
|
|
130
|
+
# `irb.rd.ja` and `doc.ja.md` put the marker in a trailing extension.
|
|
131
|
+
extra = File.basename(path).downcase.split(".").map { |p| p.sub(/^rd$/, "") }
|
|
132
|
+
extra.any? { |p| FOREIGN_MARKERS.include?(p) }
|
|
133
|
+
end
|
|
134
|
+
|
|
135
|
+
# scrub on the success path too, not only when the read raises: a file
|
|
136
|
+
# with a few invalid bytes reads fine and then explodes in gsub, which
|
|
137
|
+
# costs a gem its prose AND aborts the run part-way through the corpus.
|
|
138
|
+
def read(path)
|
|
139
|
+
File.read(path, mode: "r:UTF-8").scrub
|
|
140
|
+
rescue
|
|
141
|
+
File.binread(path).force_encoding("UTF-8").scrub
|
|
142
|
+
end
|
|
143
|
+
|
|
144
|
+
def chunks_for(gem_name, gem_version, doc)
|
|
145
|
+
Markdown.sections(clean(doc.text)).flat_map { |section|
|
|
146
|
+
body = tidy(section.body)
|
|
147
|
+
next [] if body.empty?
|
|
148
|
+
|
|
149
|
+
# Split once, then label, so the part count is known without
|
|
150
|
+
# re-splitting the body for every part's title.
|
|
151
|
+
parts = split(body, max: Chunker::MAX_BODY_CHARS)
|
|
152
|
+
total = parts.size
|
|
153
|
+
|
|
154
|
+
parts.each_with_index.map { |part, i|
|
|
155
|
+
Chunker::Chunk.new(
|
|
156
|
+
gem_name: gem_name,
|
|
157
|
+
gem_version: gem_version,
|
|
158
|
+
source_type: doc.source_type,
|
|
159
|
+
title: title(gem_name, doc, section, total: total, index: i),
|
|
160
|
+
body: part,
|
|
161
|
+
class_name: nil,
|
|
162
|
+
method_name: nil,
|
|
163
|
+
method_type: nil,
|
|
164
|
+
signature: nil
|
|
165
|
+
)
|
|
166
|
+
}
|
|
167
|
+
}
|
|
168
|
+
end
|
|
169
|
+
|
|
170
|
+
# A `· 2/5` suffix only appears once a section is actually split, so the
|
|
171
|
+
# common case keeps a clean title and a search returning three parts of one
|
|
172
|
+
# section is not indistinguishable from three unrelated hits.
|
|
173
|
+
def title(gem_name, doc, section, total:, index:)
|
|
174
|
+
base = [gem_name, doc.label, section.title].compact.reject(&:empty?).join(" › ")
|
|
175
|
+
|
|
176
|
+
(total > 1) ? "#{base} · #{index + 1}/#{total}" : base
|
|
177
|
+
end
|
|
178
|
+
|
|
179
|
+
# Split on blank lines so a chunk boundary is a paragraph boundary, never
|
|
180
|
+
# mid-sentence. Paragraphs longer than the cap are hard-cut as a last
|
|
181
|
+
# resort, because dropping them would lose content silently.
|
|
182
|
+
def split(body, max:)
|
|
183
|
+
return [body] if body.length <= max
|
|
184
|
+
|
|
185
|
+
out = []
|
|
186
|
+
current = +""
|
|
187
|
+
body.split(/\n{2,}/).each do |paragraph|
|
|
188
|
+
if current.empty?
|
|
189
|
+
current << paragraph.dup
|
|
190
|
+
elsif current.length + paragraph.length + 2 <= max
|
|
191
|
+
current << "\n\n" << paragraph
|
|
192
|
+
else
|
|
193
|
+
out << current
|
|
194
|
+
current = +""
|
|
195
|
+
current << paragraph.dup
|
|
196
|
+
end
|
|
197
|
+
end
|
|
198
|
+
out << current unless current.empty?
|
|
199
|
+
out.flat_map { |chunk| (chunk.length <= max) ? [chunk] : chunk.scan(/.{1,#{max}}/m) }
|
|
200
|
+
end
|
|
201
|
+
|
|
202
|
+
def clean(text)
|
|
203
|
+
text.gsub(/\[!\[[^\]]*\]\([^)]*\)\]\([^)]*\)/, "")
|
|
204
|
+
end
|
|
205
|
+
|
|
206
|
+
def tidy(body)
|
|
207
|
+
body.gsub(/<!--.*?-->/m, "").strip
|
|
208
|
+
end
|
|
209
|
+
end
|
|
210
|
+
end
|
|
211
|
+
end
|
data/lib/gemchat/ri.rb
ADDED
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Gemchat
|
|
4
|
+
# Generates and reads ri documentation for an installed gem.
|
|
5
|
+
#
|
|
6
|
+
# ri is the primary corpus: it carries the API surface (class, method,
|
|
7
|
+
# signature, description) that an agent needs to call a gem correctly. Only
|
|
8
|
+
# lib/ and ext/ are documented, and --all is deliberately omitted, so private
|
|
9
|
+
# and internal machinery stays out of the index.
|
|
10
|
+
class Ri
|
|
11
|
+
SCHEMA_VERSION = "1"
|
|
12
|
+
|
|
13
|
+
class GenerationError < StandardError; end
|
|
14
|
+
|
|
15
|
+
def self.documented_dirs(gem_dir)
|
|
16
|
+
%w[lib ext].select { |dir| File.directory?(File.join(gem_dir, dir)) }
|
|
17
|
+
end
|
|
18
|
+
|
|
19
|
+
# Reads an existing ri store, as installed by `gem install` or `gem rdoc`.
|
|
20
|
+
# Returns nil when absent. Bundler does not generate ri by default, so this
|
|
21
|
+
# is a fast path rather than the common case.
|
|
22
|
+
def self.existing_store_path(name, version)
|
|
23
|
+
roots = gem_roots
|
|
24
|
+
roots.each do |root|
|
|
25
|
+
candidate = File.join(root, "doc", "#{name}-#{version}", "ri")
|
|
26
|
+
return candidate if File.directory?(candidate)
|
|
27
|
+
end
|
|
28
|
+
nil
|
|
29
|
+
end
|
|
30
|
+
|
|
31
|
+
# Bundler's install root is usually absent from Gem.path. Resolved
|
|
32
|
+
# defensively: a bare lockfile with no active bundle is normal when gemchat
|
|
33
|
+
# runs outside `bundle exec`.
|
|
34
|
+
def self.gem_roots
|
|
35
|
+
roots = Gem.path.dup
|
|
36
|
+
if defined?(Bundler)
|
|
37
|
+
begin
|
|
38
|
+
bundle_path = Bundler.bundle_path.to_s
|
|
39
|
+
if File.directory?(bundle_path)
|
|
40
|
+
roots << bundle_path
|
|
41
|
+
roots << File.join(bundle_path, "ruby", RbConfig::CONFIG["ruby_version"])
|
|
42
|
+
end
|
|
43
|
+
rescue StandardError, LoadError
|
|
44
|
+
nil
|
|
45
|
+
end
|
|
46
|
+
end
|
|
47
|
+
roots.compact.uniq
|
|
48
|
+
end
|
|
49
|
+
|
|
50
|
+
def self.locate(name, version)
|
|
51
|
+
existing_store_path(name, version) || generated_store_path(name, version)
|
|
52
|
+
end
|
|
53
|
+
|
|
54
|
+
def self.generated_store_path(name, version)
|
|
55
|
+
File.join(Gemchat.root, "ri", "#{name}-#{version}")
|
|
56
|
+
end
|
|
57
|
+
|
|
58
|
+
# Generates an ri store into the private cache. RDoc creates the output
|
|
59
|
+
# directory itself and refuses to write into one that already exists, so
|
|
60
|
+
# the parent is created and the target is left alone.
|
|
61
|
+
def self.generate(name, version, gem_dir:, output_dir:)
|
|
62
|
+
dirs = documented_dirs(gem_dir)
|
|
63
|
+
return nil if dirs.empty?
|
|
64
|
+
|
|
65
|
+
FileUtils.mkdir_p(File.dirname(output_dir))
|
|
66
|
+
return output_dir if File.directory?(output_dir)
|
|
67
|
+
|
|
68
|
+
args = ["--ri", "--op", output_dir, "--quiet", "--force-update", *dirs]
|
|
69
|
+
Dir.chdir(gem_dir) { RDoc::RDoc.new.document(args) }
|
|
70
|
+
output_dir
|
|
71
|
+
rescue SystemStackError
|
|
72
|
+
# RDoc recurses on symlink cycles and deeply nested modules.
|
|
73
|
+
# SystemStackError < Exception, so a bare rescue misses it, and a
|
|
74
|
+
# truncated store must never be read.
|
|
75
|
+
FileUtils.rm_rf(output_dir)
|
|
76
|
+
raise GenerationError, "ri generation for #{name} overflowed the stack"
|
|
77
|
+
rescue => e
|
|
78
|
+
FileUtils.rm_rf(output_dir)
|
|
79
|
+
raise GenerationError, "ri generation for #{name} failed: #{e.message}"
|
|
80
|
+
end
|
|
81
|
+
|
|
82
|
+
# Reads an RDoc store into plain Ruby data:
|
|
83
|
+
# [{ class_name:, description:, methods: [{ name:, type:, signature:, description: }] }]
|
|
84
|
+
def self.read(ri_path, name)
|
|
85
|
+
return [] unless File.directory?(ri_path)
|
|
86
|
+
|
|
87
|
+
driver = RDoc::RI::Driver.new(RDoc::RI::Driver.process_args(["--doc-dir", ri_path]))
|
|
88
|
+
store = driver.stores.find { |s| s.path == ri_path || s.path.to_s.include?(name) }
|
|
89
|
+
return [] unless store
|
|
90
|
+
|
|
91
|
+
store.cache[:modules].filter_map { |class_name| read_class(store, class_name) }
|
|
92
|
+
end
|
|
93
|
+
|
|
94
|
+
def self.read_class(store, class_name)
|
|
95
|
+
klass = store.load_class(class_name)
|
|
96
|
+
methods = klass.method_list.filter_map { |m| read_method(store, class_name, m) }
|
|
97
|
+
{
|
|
98
|
+
class_name: class_name,
|
|
99
|
+
description: markup_to_text(klass.comment),
|
|
100
|
+
methods: methods
|
|
101
|
+
}
|
|
102
|
+
rescue
|
|
103
|
+
nil
|
|
104
|
+
end
|
|
105
|
+
private_class_method :read_class
|
|
106
|
+
|
|
107
|
+
def self.read_method(store, class_name, method)
|
|
108
|
+
full = store.load_method(class_name, "#{(method.type == "class") ? "::" : "#"}#{method.name}")
|
|
109
|
+
{
|
|
110
|
+
name: method.name,
|
|
111
|
+
type: method.type,
|
|
112
|
+
signature: full.arglists || full.call_seq || method.name,
|
|
113
|
+
description: markup_to_text(full.comment)
|
|
114
|
+
}
|
|
115
|
+
rescue
|
|
116
|
+
nil
|
|
117
|
+
end
|
|
118
|
+
private_class_method :read_method
|
|
119
|
+
|
|
120
|
+
def self.markup_to_text(comment)
|
|
121
|
+
return "" unless comment
|
|
122
|
+
|
|
123
|
+
doc = comment.respond_to?(:parse) ? comment.parse : comment
|
|
124
|
+
doc.accept(RDoc::Markup::ToMarkdown.new).strip
|
|
125
|
+
rescue
|
|
126
|
+
""
|
|
127
|
+
end
|
|
128
|
+
private_class_method :markup_to_text
|
|
129
|
+
end
|
|
130
|
+
end
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module Gemchat
|
|
4
|
+
# The stopwords the lexical arm drops before building an FTS5 query.
|
|
5
|
+
#
|
|
6
|
+
# FTS5 has no `stopwords=` option, so without this the words below sit in the
|
|
7
|
+
# index and match. On an OR'd query that is the whole problem: `the`, `of` and
|
|
8
|
+
# `how` appear in nearly every chunk, so they match everything and rank
|
|
9
|
+
# nothing. Measured on a ~22k-chunk corpus, the OR'd arm located the right
|
|
10
|
+
# chunk for 1 of 5 known-answer queries; see §11.3 of the plan.
|
|
11
|
+
#
|
|
12
|
+
# Vendored from PostgreSQL's `src/backend/snowball/stopwords/english.stop`,
|
|
13
|
+
# the same 127 words that `websearch_to_tsquery('english', ...)` uses in
|
|
14
|
+
# gemchat_app. Pinned by a count in the tests because a truncated paste is the
|
|
15
|
+
# most likely way this rots silently, and it would fail quietly -- the arm
|
|
16
|
+
# would just get slightly worse.
|
|
17
|
+
#
|
|
18
|
+
# Do NOT substitute gemchat_app's own list. That one
|
|
19
|
+
# (`app/services/agentic/gem_similarity_service.rb`) is tuned for SEO title
|
|
20
|
+
# work and drops `gem`, `ruby`, `library`, `tool`, `management` and `system`,
|
|
21
|
+
# all of which are signal in a documentation index.
|
|
22
|
+
#
|
|
23
|
+
# Not the same algorithm, either: this list is paired with FTS5's `porter`
|
|
24
|
+
# (Porter-1980) in the schema, whereas gemchat_app stems with Snowball via
|
|
25
|
+
# PostgreSQL. Local and hosted retrieval will never stem identically.
|
|
26
|
+
module Stopwords
|
|
27
|
+
ENGLISH = %w[
|
|
28
|
+
i me my myself we our ours ourselves
|
|
29
|
+
you your yours yourself yourselves he him his himself she her hers herself
|
|
30
|
+
it its itself they them their theirs themselves
|
|
31
|
+
what which who whom this that these those
|
|
32
|
+
am is are was were be been being have has had having
|
|
33
|
+
do does did doing a an the and but if or because as until while
|
|
34
|
+
of at by for with about against between into through during
|
|
35
|
+
before after above below to from up down
|
|
36
|
+
in out on off over under again further then once here there
|
|
37
|
+
when where why how all any both each few more most other some such
|
|
38
|
+
no nor not only own same so than too very s t
|
|
39
|
+
can will just don should now
|
|
40
|
+
].to_set.freeze
|
|
41
|
+
|
|
42
|
+
# True when every alphanumeric token of `term` is a stopword.
|
|
43
|
+
#
|
|
44
|
+
# The tokenisation matters and is not a detail. `unicode61` splits on
|
|
45
|
+
# non-alphanumerics, so `how-to` is two tokens -- `how` and `to` -- both
|
|
46
|
+
# stopwords, and `don't` is `don` and `t`, which are *both* in the upstream
|
|
47
|
+
# list precisely because Snowball wrote it that way. A plain `include?` on the
|
|
48
|
+
# whole term would keep both, which is where the noise survives.
|
|
49
|
+
#
|
|
50
|
+
# `[[:alnum:]]` rather than `[a-z0-9]` because unicode61 keeps accented and
|
|
51
|
+
# non-Latin letters inside a token; splitting on ASCII would tear `café` into
|
|
52
|
+
# `caf`, which happens to be harmless here (a fragment is not a stopword, so
|
|
53
|
+
# the term is kept) but is the wrong rule to encode.
|
|
54
|
+
#
|
|
55
|
+
# A term with no alphanumeric tokens at all is *not* treated as a stopword.
|
|
56
|
+
# Returning false keeps it in the query, so a degenerate term cannot empty
|
|
57
|
+
# the list and turn a real question into a nil query.
|
|
58
|
+
def self.only_stopwords?(term)
|
|
59
|
+
tokens = term.to_s.downcase.scan(/[[:alnum:]]+/)
|
|
60
|
+
return false if tokens.empty?
|
|
61
|
+
|
|
62
|
+
tokens.all? { |token| ENGLISH.include?(token) }
|
|
63
|
+
end
|
|
64
|
+
end
|
|
65
|
+
end
|