gemchat 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,33 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Gemchat
4
+ # Where things live on disk.
5
+ #
6
+ # Its own file so that anything loaded on its own — notably the Bundler plugin
7
+ # shim, which runs in a fresh child during `bundle install` — can resolve a path
8
+ # without requiring the whole gem, and without a circular require back through
9
+ # lib/gemchat.rb.
10
+ module Paths
11
+ module_function
12
+
13
+ # Index location: the ri cache and index.sqlite.
14
+ #
15
+ # GEMCHAT_HOME is taken verbatim. An override that silently grew a suffix
16
+ # would make it impossible to predict where the data went.
17
+ #
18
+ # Deliberately not memoised. Memoising a value derived from the environment
19
+ # means the first reader wins for the life of the process, so setting
20
+ # GEMCHAT_HOME later does nothing — which once made the test suite write to
21
+ # the developer's real ~/.gemchat. The join is cheap enough to repeat.
22
+ def root
23
+ File.expand_path(ENV["GEMCHAT_HOME"] || File.join(Dir.home, ".gemchat"))
24
+ end
25
+
26
+ # Settings and trust records, kept separate from the index so wiping the
27
+ # cache does not also discard the decisions about which projects to trust.
28
+ def config_home
29
+ ENV["GEMCHAT_CONFIG_HOME"] ||
30
+ File.join(ENV["XDG_CONFIG_HOME"] || File.join(Dir.home, ".config"), "gemchat")
31
+ end
32
+ end
33
+ end
@@ -0,0 +1,63 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "yaml"
4
+
5
+ module Gemchat
6
+ # Reads Bundler's own record of which plugins are installed and which hooks
7
+ # they registered.
8
+ #
9
+ # This exists because "is the hook actually armed?" cannot be answered by
10
+ # looking at the Gemfile. A Gemfile line only says the plugin was *requested*;
11
+ # Bundler activates it on the next install, and only writes this index when it
12
+ # does. The difference is the whole point — §15.1a lists four ways a plugin
13
+ # looks installed and does nothing, and `hooks:` being empty here is the
14
+ # signature of the worst one: a plugin that ran once, during the install that
15
+ # installed it, and never again.
16
+ module PluginIndex
17
+ FILENAME = "index"
18
+
19
+ module_function
20
+
21
+ # Honours BUNDLE_APP_CONFIG the way Bundler does, so this reads the same
22
+ # index the hook does.
23
+ def dir
24
+ config = ENV["BUNDLE_APP_CONFIG"]
25
+ config ||= File.join(Dir.pwd, ".bundle")
26
+ File.join(config, "plugin")
27
+ end
28
+
29
+ def path
30
+ File.join(dir, FILENAME)
31
+ end
32
+
33
+ def exist?
34
+ File.file?(path)
35
+ end
36
+
37
+ def data
38
+ return nil unless exist?
39
+
40
+ parsed = YAML.safe_load_file(path, permitted_classes: [Symbol], aliases: true)
41
+ parsed.is_a?(Hash) ? parsed : nil
42
+ rescue
43
+ nil
44
+ end
45
+
46
+ def installed_plugins
47
+ data&.dig("plugin_paths")&.keys || []
48
+ end
49
+
50
+ def registered_hooks
51
+ data&.dig("hooks") || {}
52
+ end
53
+
54
+ def hook?(name, event = "after-install-all")
55
+ Array(registered_hooks[event]).include?(name)
56
+ end
57
+
58
+ # Whether the hook will actually run, as opposed to being merely declared.
59
+ def armed?(name = "gemchat", event: "after-install-all")
60
+ hook?(name, event)
61
+ end
62
+ end
63
+ end
@@ -0,0 +1,211 @@
1
+ # frozen_string_literal: true
2
+
3
+ require_relative "markdown"
4
+
5
+ module Gemchat
6
+ # The prose half of a gem: its README, its guides, and the reference material
7
+ # it ships under `doc/`.
8
+ #
9
+ # ri answers "what does this method take and return". Prose answers "how is
10
+ # this library actually used", which is the question an agent most often has
11
+ # and the one ri is worst at. Each source is stored under its own
12
+ # `source_type`, so a gem's README and its reference manual never overwrite one
13
+ # another and search output can say what kind of document a hit came from.
14
+ #
15
+ # The `readme` / `guides` / `doc` split is not incidental: `gemchat_app` already
16
+ # uses exactly this vocabulary, gives `guides` a ranking prior of 1.15, and
17
+ # groups `readme`+`guides` only at query time to enforce prose diversity. See
18
+ # §20.1 of the plan.
19
+ class Prose
20
+ FILENAME = "README"
21
+ PROSE_EXTENSIONS = %w[.md .markdown .mkd .rdoc .txt .textile].freeze
22
+ GUIDE_DIRS = %w[guides guide].freeze
23
+ DOC_DIRS = %w[doc docs].freeze
24
+
25
+ # Generated output. A gem installed with documentation has ri under
26
+ # `doc/<name>-<version>/ri`, and older gemspecs committed rdoc's HTML tree.
27
+ # Both duplicate what the ri half already covers, so both are dropped — but
28
+ # only from `doc/`, and only by shape. Hand-written files are kept.
29
+ GENERATED_DIRS = %w[ri].freeze
30
+ GENERATED_EXTENSIONS = %w[.html .htm .js .css].freeze
31
+
32
+ # Languages seen in bundled documentation, keyed by the marker that appears
33
+ # in a filename or directory name. Skipped by default: they cost embedding
34
+ # time and can never match an English query, while polluting BM25 with tokens
35
+ # no English query contains. `en` is deliberately absent.
36
+ FOREIGN_MARKERS = %w[
37
+ ja zh ko fr de es pt it ru pl tr nl sv da no fi cs hu el he ar hi th vi
38
+ id uk ro bg sr hr sk sl et lv lt ca gl eu af sw fa ur bn ta ms tl
39
+ ].freeze
40
+
41
+ Document = Struct.new(:source_type, :label, :path, :text)
42
+
43
+ class << self
44
+ def chunks(gem_name, gem_version, gem_dir, all_languages: false)
45
+ documents(gem_dir, all_languages: all_languages)
46
+ .flat_map { |doc| chunks_for(gem_name, gem_version, doc) }
47
+ end
48
+
49
+ def documents(gem_dir, all_languages: false)
50
+ paths(gem_dir, all_languages: all_languages).map { |path|
51
+ Document.new(source_type_for(gem_dir, path), label_for(gem_dir, path),
52
+ path, read(path))
53
+ }
54
+ end
55
+
56
+ def paths(gem_dir, all_languages: false)
57
+ return [] unless File.directory?(gem_dir)
58
+
59
+ all = [readme_in(gem_dir)] +
60
+ within(gem_dir, GUIDE_DIRS) +
61
+ within(gem_dir, DOC_DIRS)
62
+
63
+ all.compact.uniq.select { |path| indexable?(path, all_languages) }
64
+ end
65
+
66
+ def source_type_for(gem_dir, path)
67
+ relative = path.sub("#{gem_dir}/", "")
68
+ return "readme" unless relative.include?("/")
69
+
70
+ first = relative.split("/").first
71
+ GUIDE_DIRS.include?(first) ? "guides" : "doc"
72
+ end
73
+
74
+ # The directory the document lives in, relative to the gem root. Keeps
75
+ # `doc/rexml/parsers.md` and `doc/csv/io.rdoc` distinguishable in a result
76
+ # title, which "doc" alone would not.
77
+ def label_for(gem_dir, path)
78
+ relative = path.sub("#{gem_dir}/", "")
79
+ return "README" unless relative.include?("/")
80
+
81
+ File.dirname(relative)
82
+ end
83
+
84
+ private
85
+
86
+ def within(gem_dir, dirs)
87
+ dirs.flat_map { |d| Dir.glob(File.join(gem_dir, d, "**", "*")) }
88
+ end
89
+
90
+ def readme_in(gem_dir)
91
+ Prose::PROSE_EXTENSIONS.each do |ext|
92
+ candidate = File.join(gem_dir, "#{FILENAME}#{ext}")
93
+ return candidate if File.file?(candidate)
94
+ end
95
+ bare = File.join(gem_dir, FILENAME)
96
+ bare if File.file?(bare)
97
+ end
98
+
99
+ def indexable?(path, all_languages)
100
+ return false unless File.file?(path)
101
+ return false if foreign?(path) && !all_languages
102
+ # A README is taken whatever its extension; everything under a guide or
103
+ # doc directory has to look like prose, or the tree fills up with
104
+ # example sources and translation catalogues.
105
+ return true if File.basename(path).start_with?(FILENAME.downcase)
106
+ return false unless PROSE_EXTENSIONS.include?(extension(path))
107
+
108
+ !generated?(path)
109
+ end
110
+
111
+ def extension(path)
112
+ File.extname(path).downcase
113
+ end
114
+
115
+ def generated?(path)
116
+ parts = path.split("/")
117
+ return true if parts.any? { |p| GENERATED_DIRS.include?(p) }
118
+
119
+ GENERATED_EXTENSIONS.include?(extension(path))
120
+ end
121
+
122
+ # Matches a language marker only when it stands alone as a path segment or
123
+ # as a dot- or underscore-delimited token in the basename, so a directory
124
+ # called `text` or `example` is not mistaken for one.
125
+ def foreign?(path)
126
+ parts = path.split("/")
127
+ base = File.basename(parts.pop, ".*").downcase
128
+ tokens = base.split(/[._-]/).map { |p| p.sub(/\.(rd|md|txt|textile)$/, "") }
129
+ return true if (parts + tokens).any? { |p| FOREIGN_MARKERS.include?(p.downcase) }
130
+ # `irb.rd.ja` and `doc.ja.md` put the marker in a trailing extension.
131
+ extra = File.basename(path).downcase.split(".").map { |p| p.sub(/^rd$/, "") }
132
+ extra.any? { |p| FOREIGN_MARKERS.include?(p) }
133
+ end
134
+
135
+ # scrub on the success path too, not only when the read raises: a file
136
+ # with a few invalid bytes reads fine and then explodes in gsub, which
137
+ # costs a gem its prose AND aborts the run part-way through the corpus.
138
+ def read(path)
139
+ File.read(path, mode: "r:UTF-8").scrub
140
+ rescue
141
+ File.binread(path).force_encoding("UTF-8").scrub
142
+ end
143
+
144
+ def chunks_for(gem_name, gem_version, doc)
145
+ Markdown.sections(clean(doc.text)).flat_map { |section|
146
+ body = tidy(section.body)
147
+ next [] if body.empty?
148
+
149
+ # Split once, then label, so the part count is known without
150
+ # re-splitting the body for every part's title.
151
+ parts = split(body, max: Chunker::MAX_BODY_CHARS)
152
+ total = parts.size
153
+
154
+ parts.each_with_index.map { |part, i|
155
+ Chunker::Chunk.new(
156
+ gem_name: gem_name,
157
+ gem_version: gem_version,
158
+ source_type: doc.source_type,
159
+ title: title(gem_name, doc, section, total: total, index: i),
160
+ body: part,
161
+ class_name: nil,
162
+ method_name: nil,
163
+ method_type: nil,
164
+ signature: nil
165
+ )
166
+ }
167
+ }
168
+ end
169
+
170
+ # A `· 2/5` suffix only appears once a section is actually split, so the
171
+ # common case keeps a clean title and a search returning three parts of one
172
+ # section is not indistinguishable from three unrelated hits.
173
+ def title(gem_name, doc, section, total:, index:)
174
+ base = [gem_name, doc.label, section.title].compact.reject(&:empty?).join(" › ")
175
+
176
+ (total > 1) ? "#{base} · #{index + 1}/#{total}" : base
177
+ end
178
+
179
+ # Split on blank lines so a chunk boundary is a paragraph boundary, never
180
+ # mid-sentence. Paragraphs longer than the cap are hard-cut as a last
181
+ # resort, because dropping them would lose content silently.
182
+ def split(body, max:)
183
+ return [body] if body.length <= max
184
+
185
+ out = []
186
+ current = +""
187
+ body.split(/\n{2,}/).each do |paragraph|
188
+ if current.empty?
189
+ current << paragraph.dup
190
+ elsif current.length + paragraph.length + 2 <= max
191
+ current << "\n\n" << paragraph
192
+ else
193
+ out << current
194
+ current = +""
195
+ current << paragraph.dup
196
+ end
197
+ end
198
+ out << current unless current.empty?
199
+ out.flat_map { |chunk| (chunk.length <= max) ? [chunk] : chunk.scan(/.{1,#{max}}/m) }
200
+ end
201
+
202
+ def clean(text)
203
+ text.gsub(/\[!\[[^\]]*\]\([^)]*\)\]\([^)]*\)/, "")
204
+ end
205
+
206
+ def tidy(body)
207
+ body.gsub(/<!--.*?-->/m, "").strip
208
+ end
209
+ end
210
+ end
211
+ end
data/lib/gemchat/ri.rb ADDED
@@ -0,0 +1,130 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Gemchat
4
+ # Generates and reads ri documentation for an installed gem.
5
+ #
6
+ # ri is the primary corpus: it carries the API surface (class, method,
7
+ # signature, description) that an agent needs to call a gem correctly. Only
8
+ # lib/ and ext/ are documented, and --all is deliberately omitted, so private
9
+ # and internal machinery stays out of the index.
10
+ class Ri
11
+ SCHEMA_VERSION = "1"
12
+
13
+ class GenerationError < StandardError; end
14
+
15
+ def self.documented_dirs(gem_dir)
16
+ %w[lib ext].select { |dir| File.directory?(File.join(gem_dir, dir)) }
17
+ end
18
+
19
+ # Reads an existing ri store, as installed by `gem install` or `gem rdoc`.
20
+ # Returns nil when absent. Bundler does not generate ri by default, so this
21
+ # is a fast path rather than the common case.
22
+ def self.existing_store_path(name, version)
23
+ roots = gem_roots
24
+ roots.each do |root|
25
+ candidate = File.join(root, "doc", "#{name}-#{version}", "ri")
26
+ return candidate if File.directory?(candidate)
27
+ end
28
+ nil
29
+ end
30
+
31
+ # Bundler's install root is usually absent from Gem.path. Resolved
32
+ # defensively: a bare lockfile with no active bundle is normal when gemchat
33
+ # runs outside `bundle exec`.
34
+ def self.gem_roots
35
+ roots = Gem.path.dup
36
+ if defined?(Bundler)
37
+ begin
38
+ bundle_path = Bundler.bundle_path.to_s
39
+ if File.directory?(bundle_path)
40
+ roots << bundle_path
41
+ roots << File.join(bundle_path, "ruby", RbConfig::CONFIG["ruby_version"])
42
+ end
43
+ rescue StandardError, LoadError
44
+ nil
45
+ end
46
+ end
47
+ roots.compact.uniq
48
+ end
49
+
50
+ def self.locate(name, version)
51
+ existing_store_path(name, version) || generated_store_path(name, version)
52
+ end
53
+
54
+ def self.generated_store_path(name, version)
55
+ File.join(Gemchat.root, "ri", "#{name}-#{version}")
56
+ end
57
+
58
+ # Generates an ri store into the private cache. RDoc creates the output
59
+ # directory itself and refuses to write into one that already exists, so
60
+ # the parent is created and the target is left alone.
61
+ def self.generate(name, version, gem_dir:, output_dir:)
62
+ dirs = documented_dirs(gem_dir)
63
+ return nil if dirs.empty?
64
+
65
+ FileUtils.mkdir_p(File.dirname(output_dir))
66
+ return output_dir if File.directory?(output_dir)
67
+
68
+ args = ["--ri", "--op", output_dir, "--quiet", "--force-update", *dirs]
69
+ Dir.chdir(gem_dir) { RDoc::RDoc.new.document(args) }
70
+ output_dir
71
+ rescue SystemStackError
72
+ # RDoc recurses on symlink cycles and deeply nested modules.
73
+ # SystemStackError < Exception, so a bare rescue misses it, and a
74
+ # truncated store must never be read.
75
+ FileUtils.rm_rf(output_dir)
76
+ raise GenerationError, "ri generation for #{name} overflowed the stack"
77
+ rescue => e
78
+ FileUtils.rm_rf(output_dir)
79
+ raise GenerationError, "ri generation for #{name} failed: #{e.message}"
80
+ end
81
+
82
+ # Reads an RDoc store into plain Ruby data:
83
+ # [{ class_name:, description:, methods: [{ name:, type:, signature:, description: }] }]
84
+ def self.read(ri_path, name)
85
+ return [] unless File.directory?(ri_path)
86
+
87
+ driver = RDoc::RI::Driver.new(RDoc::RI::Driver.process_args(["--doc-dir", ri_path]))
88
+ store = driver.stores.find { |s| s.path == ri_path || s.path.to_s.include?(name) }
89
+ return [] unless store
90
+
91
+ store.cache[:modules].filter_map { |class_name| read_class(store, class_name) }
92
+ end
93
+
94
+ def self.read_class(store, class_name)
95
+ klass = store.load_class(class_name)
96
+ methods = klass.method_list.filter_map { |m| read_method(store, class_name, m) }
97
+ {
98
+ class_name: class_name,
99
+ description: markup_to_text(klass.comment),
100
+ methods: methods
101
+ }
102
+ rescue
103
+ nil
104
+ end
105
+ private_class_method :read_class
106
+
107
+ def self.read_method(store, class_name, method)
108
+ full = store.load_method(class_name, "#{(method.type == "class") ? "::" : "#"}#{method.name}")
109
+ {
110
+ name: method.name,
111
+ type: method.type,
112
+ signature: full.arglists || full.call_seq || method.name,
113
+ description: markup_to_text(full.comment)
114
+ }
115
+ rescue
116
+ nil
117
+ end
118
+ private_class_method :read_method
119
+
120
+ def self.markup_to_text(comment)
121
+ return "" unless comment
122
+
123
+ doc = comment.respond_to?(:parse) ? comment.parse : comment
124
+ doc.accept(RDoc::Markup::ToMarkdown.new).strip
125
+ rescue
126
+ ""
127
+ end
128
+ private_class_method :markup_to_text
129
+ end
130
+ end
@@ -0,0 +1,65 @@
1
+ # frozen_string_literal: true
2
+
3
+ module Gemchat
4
+ # The stopwords the lexical arm drops before building an FTS5 query.
5
+ #
6
+ # FTS5 has no `stopwords=` option, so without this the words below sit in the
7
+ # index and match. On an OR'd query that is the whole problem: `the`, `of` and
8
+ # `how` appear in nearly every chunk, so they match everything and rank
9
+ # nothing. Measured on a ~22k-chunk corpus, the OR'd arm located the right
10
+ # chunk for 1 of 5 known-answer queries; see §11.3 of the plan.
11
+ #
12
+ # Vendored from PostgreSQL's `src/backend/snowball/stopwords/english.stop`,
13
+ # the same 127 words that `websearch_to_tsquery('english', ...)` uses in
14
+ # gemchat_app. Pinned by a count in the tests because a truncated paste is the
15
+ # most likely way this rots silently, and it would fail quietly -- the arm
16
+ # would just get slightly worse.
17
+ #
18
+ # Do NOT substitute gemchat_app's own list. That one
19
+ # (`app/services/agentic/gem_similarity_service.rb`) is tuned for SEO title
20
+ # work and drops `gem`, `ruby`, `library`, `tool`, `management` and `system`,
21
+ # all of which are signal in a documentation index.
22
+ #
23
+ # Not the same algorithm, either: this list is paired with FTS5's `porter`
24
+ # (Porter-1980) in the schema, whereas gemchat_app stems with Snowball via
25
+ # PostgreSQL. Local and hosted retrieval will never stem identically.
26
+ module Stopwords
27
+ ENGLISH = %w[
28
+ i me my myself we our ours ourselves
29
+ you your yours yourself yourselves he him his himself she her hers herself
30
+ it its itself they them their theirs themselves
31
+ what which who whom this that these those
32
+ am is are was were be been being have has had having
33
+ do does did doing a an the and but if or because as until while
34
+ of at by for with about against between into through during
35
+ before after above below to from up down
36
+ in out on off over under again further then once here there
37
+ when where why how all any both each few more most other some such
38
+ no nor not only own same so than too very s t
39
+ can will just don should now
40
+ ].to_set.freeze
41
+
42
+ # True when every alphanumeric token of `term` is a stopword.
43
+ #
44
+ # The tokenisation matters and is not a detail. `unicode61` splits on
45
+ # non-alphanumerics, so `how-to` is two tokens -- `how` and `to` -- both
46
+ # stopwords, and `don't` is `don` and `t`, which are *both* in the upstream
47
+ # list precisely because Snowball wrote it that way. A plain `include?` on the
48
+ # whole term would keep both, which is where the noise survives.
49
+ #
50
+ # `[[:alnum:]]` rather than `[a-z0-9]` because unicode61 keeps accented and
51
+ # non-Latin letters inside a token; splitting on ASCII would tear `café` into
52
+ # `caf`, which happens to be harmless here (a fragment is not a stopword, so
53
+ # the term is kept) but is the wrong rule to encode.
54
+ #
55
+ # A term with no alphanumeric tokens at all is *not* treated as a stopword.
56
+ # Returning false keeps it in the query, so a degenerate term cannot empty
57
+ # the list and turn a real question into a nil query.
58
+ def self.only_stopwords?(term)
59
+ tokens = term.to_s.downcase.scan(/[[:alnum:]]+/)
60
+ return false if tokens.empty?
61
+
62
+ tokens.all? { |token| ENGLISH.include?(token) }
63
+ end
64
+ end
65
+ end