portage-cli 0.7.5 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (82) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +616 -0
  3. data/README.md +302 -5
  4. data/known-stores/categories.yml +1263 -0
  5. data/lib/portage/cli/agent_profile_url.rb +30 -0
  6. data/lib/portage/cli/approval_policy.rb +56 -0
  7. data/lib/portage/cli/approve.rb +135 -0
  8. data/lib/portage/cli/browser_import/categorize.rb +59 -0
  9. data/lib/portage/cli/browser_import/confirm.rb +35 -0
  10. data/lib/portage/cli/browser_import/domains.rb +47 -0
  11. data/lib/portage/cli/browser_import/filter.rb +91 -0
  12. data/lib/portage/cli/browser_import/importer.rb +248 -0
  13. data/lib/portage/cli/browser_import/plist_xml.rb +72 -0
  14. data/lib/portage/cli/browser_import/prober.rb +60 -0
  15. data/lib/portage/cli/browser_import/profiles.rb +114 -0
  16. data/lib/portage/cli/browser_import/readers.rb +179 -0
  17. data/lib/portage/cli/browser_import/saver.rb +62 -0
  18. data/lib/portage/cli/browser_import/sqlite.rb +68 -0
  19. data/lib/portage/cli/browser_import.rb +23 -0
  20. data/lib/portage/cli/browser_opener.rb +38 -0
  21. data/lib/portage/cli/browser_profile/allowlist.rb +40 -0
  22. data/lib/portage/cli/browser_profile/bridge.rb +120 -0
  23. data/lib/portage/cli/browser_profile/browsers.rb +69 -0
  24. data/lib/portage/cli/browser_profile/cdp.rb +67 -0
  25. data/lib/portage/cli/browser_profile/cdp_socket.rb +186 -0
  26. data/lib/portage/cli/browser_profile/errors.rb +26 -0
  27. data/lib/portage/cli/browser_profile/launcher.rb +34 -0
  28. data/lib/portage/cli/browser_profile/profile.rb +93 -0
  29. data/lib/portage/cli/browser_profile.rb +25 -0
  30. data/lib/portage/cli/buy.rb +559 -29
  31. data/lib/portage/cli/checkout_handoff.rb +5 -24
  32. data/lib/portage/cli/classifier.rb +158 -0
  33. data/lib/portage/cli/compare.rb +3 -0
  34. data/lib/portage/cli/doctor.rb +155 -1
  35. data/lib/portage/cli/dot_env.rb +55 -0
  36. data/lib/portage/cli/find.rb +103 -12
  37. data/lib/portage/cli/handoff_agents.rb +186 -0
  38. data/lib/portage/cli/handoff_only.rb +94 -0
  39. data/lib/portage/cli/handoff_reconciler.rb +15 -1
  40. data/lib/portage/cli/handoff_target.rb +61 -0
  41. data/lib/portage/cli/history.rb +47 -2
  42. data/lib/portage/cli/human_prompt.rb +118 -0
  43. data/lib/portage/cli/index/builder.rb +335 -0
  44. data/lib/portage/cli/index/exporter.rb +91 -0
  45. data/lib/portage/cli/index/known_cache.rb +155 -0
  46. data/lib/portage/cli/index/product_store.rb +101 -0
  47. data/lib/portage/cli/index/sources/browser.rb +31 -0
  48. data/lib/portage/cli/index/sources/shopify_catalog.rb +82 -0
  49. data/lib/portage/cli/index/sources/stores_file.rb +58 -0
  50. data/lib/portage/cli/index/sources/webmcp_sweep.rb +29 -0
  51. data/lib/portage/cli/index/sources/wikidata.rb +95 -0
  52. data/lib/portage/cli/index/sources.rb +44 -0
  53. data/lib/portage/cli/index/store.rb +109 -0
  54. data/lib/portage/cli/index.rb +20 -0
  55. data/lib/portage/cli/known_stores_url.rb +15 -0
  56. data/lib/portage/cli/money.rb +18 -0
  57. data/lib/portage/cli/offer_choice.rb +38 -0
  58. data/lib/portage/cli/offer_sources.rb +460 -0
  59. data/lib/portage/cli/payment_methods.rb +24 -3
  60. data/lib/portage/cli/pick.rb +167 -0
  61. data/lib/portage/cli/product_page.rb +84 -0
  62. data/lib/portage/cli/quotes.rb +82 -0
  63. data/lib/portage/cli/search_backends.rb +337 -12
  64. data/lib/portage/cli/setup_wizard/prompt.rb +67 -0
  65. data/lib/portage/cli/setup_wizard/steps/agent_profile.rb +60 -0
  66. data/lib/portage/cli/setup_wizard/steps/browser_import.rb +29 -0
  67. data/lib/portage/cli/setup_wizard/steps/handoff.rb +100 -0
  68. data/lib/portage/cli/setup_wizard/steps/index_build.rb +31 -0
  69. data/lib/portage/cli/setup_wizard/steps/policy.rb +60 -0
  70. data/lib/portage/cli/setup_wizard/steps/retailer_keys.rb +55 -0
  71. data/lib/portage/cli/setup_wizard/steps/search_keys.rb +54 -0
  72. data/lib/portage/cli/setup_wizard/steps/shipping.rb +51 -0
  73. data/lib/portage/cli/setup_wizard.rb +74 -0
  74. data/lib/portage/cli/version.rb +1 -1
  75. data/lib/portage/cli/webmcp.rb +10 -3
  76. data/lib/portage/cli/webmcp_autofill_confirm.rb +38 -0
  77. data/lib/portage/cli/webmcp_autofill_fields.rb +60 -0
  78. data/lib/portage/cli/webmcp_autofill_mode.rb +39 -0
  79. data/lib/portage/cli/webmcp_mapping_confirm.rb +68 -0
  80. data/lib/portage/cli/webmcp_mappings.rb +84 -0
  81. data/lib/portage/cli.rb +930 -43
  82. metadata +67 -2
@@ -0,0 +1,84 @@
1
+ require "uri"
2
+ require_relative "browser_opener"
3
+
4
+ module Portage
5
+ module Cli
6
+ # Opens the store's page for an offer or quote, so the person can look
7
+ # before they pick or approve (docs/plans/human-pick-and-approve.md
8
+ # Phase 2, "View the product page"). Used by `pick --view`, `approve
9
+ # --view` and the `v`/`v N` answers at a tty prompt. Never gated by the
10
+ # auto-open setting: the person asked for this page by name.
11
+ #
12
+ # The URL came from the store, so it's untrusted: only an http(s) URL on
13
+ # the offer's own store host is opened, and anything else is refused
14
+ # with `view_refused` rather than handed to the OS opener.
15
+ #
16
+ # Host rule: the URL's host must equal the store's host, compared
17
+ # case-insensitively with one leading `www.` ignored on either side
18
+ # (`shop.example` and `www.shop.example` are the same shop). Any other
19
+ # subdomain (`cdn.shop.example`, `shop.example.evil.test`) is refused —
20
+ # a store that keeps product pages on another host just can't be
21
+ # viewed from here. A URL carrying credentials (`user@host`) is refused
22
+ # too.
23
+ class ProductPage
24
+ include BrowserOpener
25
+
26
+ # @param url [String, nil] the product page as find/compare returned it.
27
+ # @param store [String] the offer's store origin (or a bare host).
28
+ def initialize(url:, store:)
29
+ @url = url
30
+ @store = store
31
+ end
32
+
33
+ # @return [Hash] `outcome: "viewed"` (with `opened:`) or
34
+ # `outcome: "view_refused"` (with the reason as `message:`).
35
+ def open
36
+ reason = refusal
37
+ return { outcome: "view_refused", url: @url, store: @store, message: reason } if reason
38
+
39
+ opened = open_browser(@url)
40
+ { outcome: "viewed", url: @url, opened: opened,
41
+ message: opened ? "Opened #{@url}." : "Couldn't open a browser — the page is #{@url}" }
42
+ end
43
+
44
+ # @return [String, nil] why the page can't be opened, nil when it can.
45
+ def refusal
46
+ return "No product page on record for this offer." if @url.to_s.strip.empty?
47
+
48
+ uri = web_uri(@url)
49
+ return "Not an http(s) URL: #{@url}" unless uri
50
+ return "Refusing a URL with credentials in it: #{@url}" if uri.userinfo
51
+ return nil if self.class.same_shop?(uri.host, store_host)
52
+
53
+ "#{uri.host} isn't the offer's store (#{store_host || @store}) — not opening #{@url}"
54
+ end
55
+
56
+ def self.same_shop?(host, store_host)
57
+ return false unless host && store_host
58
+
59
+ host_key(host) == host_key(store_host)
60
+ end
61
+
62
+ def self.host_key(host) = host.downcase.delete_prefix("www.")
63
+ private_class_method :host_key
64
+
65
+ private
66
+
67
+ def store_host
68
+ raw = @store.to_s.strip
69
+ parse(raw.match?(%r{\Ahttps?://}i) ? raw : "https://#{raw}")&.host
70
+ end
71
+
72
+ def web_uri(url)
73
+ uri = parse(url)
74
+ uri if uri&.host && %w[http https].include?(uri.scheme&.downcase)
75
+ end
76
+
77
+ def parse(url)
78
+ URI.parse(url.to_s.strip)
79
+ rescue URI::InvalidURIError
80
+ nil
81
+ end
82
+ end
83
+ end
84
+ end
@@ -0,0 +1,82 @@
1
+ require "json"
2
+ require "fileutils"
3
+ require "securerandom"
4
+
5
+ module Portage
6
+ module Cli
7
+ # Local record of the priced checkouts `portage buy --dry-run` has shown,
8
+ # one JSON file per quote under ~/.portage/quotes/. A quote is what
9
+ # `buy --quote QUOTE_ID --yes` later buys: it pins the store, product,
10
+ # quantity and the total that was shown, so the run that charges can
11
+ # refuse if the price has moved since (see Cli.buy_from_quote).
12
+ #
13
+ # Quotes never expire. Each is single use: #consume stamps `used_at`
14
+ # rather than deleting the file, so a spent quote is still told apart
15
+ # from one that never existed.
16
+ class Quotes
17
+ DIR = File.join(Dir.home, ".portage", "quotes").freeze
18
+ ID_FORMAT = /\Aqt_[0-9a-f]{12}\z/
19
+
20
+ def initialize(dir: DIR, now: Time.now)
21
+ @dir = dir
22
+ @now = now.to_i
23
+ end
24
+
25
+ # @param total [Integer, nil] minor units of `currency`.
26
+ # @param title [String, nil] what the checkout holds, for `portage
27
+ # approve`'s summary.
28
+ # @param url [String, nil] the offer's product page, for `approve --view`.
29
+ # @return [Hash, nil] the saved quote (string keys, `quote_id` set), or
30
+ # nil when it couldn't be written — a quote that can't be saved just
31
+ # isn't offered, never a failed dry run.
32
+ def create(store:, product_id:, qty:, total:, currency:, query: nil, offer_ref: nil, title: nil, url: nil)
33
+ quote = { "quote_id" => "qt_#{SecureRandom.hex(6)}", "offer_ref" => offer_ref, "store" => store,
34
+ "product_id" => product_id, "query" => query, "qty" => qty, "total" => total,
35
+ "currency" => currency, "title" => title, "url" => url, "created_at" => @now,
36
+ "approved" => false }
37
+ write(quote)
38
+ end
39
+
40
+ # @return [Hash, nil] nil for an unknown (or malformed) id.
41
+ def find(quote_id)
42
+ return nil unless quote_id.to_s.match?(ID_FORMAT)
43
+
44
+ parsed = JSON.parse(File.read(path(quote_id)))
45
+ parsed if parsed.is_a?(Hash)
46
+ rescue StandardError
47
+ nil
48
+ end
49
+
50
+ # Records the person's yes to this quote's total
51
+ # (docs/plans/human-pick-and-approve.md Phase 2): `by` is "person" for
52
+ # a yes typed at the tty, "agent_relayed" for one an agent passed on.
53
+ # A relayed yes never downgrades a quote the person already approved.
54
+ # @return [Hash, nil] the updated quote, nil when unknown or unwritable.
55
+ def approve(quote_id, by:)
56
+ quote = find(quote_id)
57
+ return nil unless quote
58
+ return quote if quote["approved_by"] == "person"
59
+
60
+ write(quote.merge("approved" => true, "approved_by" => by, "approved_at" => @now))
61
+ end
62
+
63
+ # Marks the quote spent. Best-effort, like every other local record.
64
+ def consume(quote_id)
65
+ quote = find(quote_id)
66
+ write(quote.merge("used_at" => @now)) if quote
67
+ end
68
+
69
+ private
70
+
71
+ def path(quote_id) = File.join(@dir, "#{quote_id}.json")
72
+
73
+ def write(quote)
74
+ FileUtils.mkdir_p(@dir)
75
+ File.write(path(quote["quote_id"]), JSON.generate(quote))
76
+ quote
77
+ rescue StandardError
78
+ nil
79
+ end
80
+ end
81
+ end
82
+ end
@@ -5,6 +5,10 @@ require "yaml"
5
5
  require "portage/ucp"
6
6
  require "portage/ucp/support/connection"
7
7
  require_relative "user_agent"
8
+ require_relative "classifier"
9
+ require_relative "index/store"
10
+ require_relative "index/product_store"
11
+ require_relative "index/known_cache"
8
12
 
9
13
  module Portage
10
14
  module Cli
@@ -31,12 +35,23 @@ module Portage
31
35
  ].freeze
32
36
 
33
37
  # Ordered cheapest/most-trusted first: your own allowlist costs no
34
- # network call and needs no key, DuckDuckGo needs no key, the keyed
35
- # engines only participate when their credentials are actually present.
38
+ # network call and needs no key, the local index next (still no
39
+ # network call, but untrusted — see Index's own comment), DuckDuckGo
40
+ # needs no key, the keyed engines only participate when their
41
+ # credentials are actually present.
36
42
  def self.default
37
- [Allowlist.new, DuckDuckGo.new, Brave.new, GoogleCse.new].select(&:available?)
43
+ [Allowlist.new, Index.new, DuckDuckGo.new, Brave.new, GoogleCse.new].select(&:available?)
38
44
  end
39
45
 
46
+ # True when the only thing standing between the caller and a real web
47
+ # search is a missing API key — i.e. DuckDuckGo's entity-only Instant
48
+ # Answer API is running solo, with no allowlist and no keyed backend
49
+ # (Brave/Google CSE) to cover the open-ended queries it can't answer.
50
+ # Used to decide whether "no candidates" is worth a nudge to set
51
+ # BRAVE_SEARCH_API_KEY / GOOGLE_CSE_KEY+GOOGLE_CSE_CX (see Find and
52
+ # Doctor) rather than a plain "nothing found".
53
+ def self.only_duckduckgo?(backends) = backends.map(&:name) == ["duckduckgo"]
54
+
40
55
  def self.get_json(uri, params: {}, headers: {})
41
56
  uri = uri.dup
42
57
  uri.query = URI.encode_www_form(params) unless params.empty?
@@ -63,14 +78,33 @@ module Portage
63
78
  end
64
79
 
65
80
  # Stores you've already decided you trust, listed in
66
- # `~/.portage/stores.yml` (a bare YAML array of URLs) or `PORTAGE_STORES`
67
- # (comma-separated — the PATH-style colon can't separate values that
68
- # contain `https://`). Query-independent on purpose: the point of the file
69
- # is "always consider these", and the store's own catalog search is what
70
- # decides whether it stocks the thing.
81
+ # `~/.portage/stores.yml` (a bare YAML array of URLs, or an array
82
+ # mixing in `{url:, categories: [...]}` entries once you've tagged
83
+ # some) or `PORTAGE_STORES` (comma-separated — the PATH-style colon
84
+ # can't separate values that contain `https://`).
85
+ #
86
+ # Trust stays query-independent: every entry is *always* a candidate
87
+ # to `find`, on the same footing whether it's tagged or not, and the
88
+ # store's own catalog search is still what decides whether it stocks
89
+ # the thing. `categories:` only changes which of these already-trusted
90
+ # entries get spent on *this* query's dozen probe slots — a stores.yml
91
+ # with fifty tagged stores across a dozen categories used to crowd out
92
+ # web-search candidates on every single query; #search now routes by
93
+ # the query's own category instead of just taking the first N.
71
94
  class Allowlist
72
95
  PATH = File.join(Dir.home, ".portage", "stores.yml").freeze
73
96
 
97
+ # At most this many of a matching category's tagged stores per
98
+ # `#search` call, so one heavily-tagged category can't fill every
99
+ # probe slot by itself.
100
+ PER_CATEGORY_CAP = 3
101
+
102
+ # The hard ceiling regardless of the caller's own `limit:` — mirrors
103
+ # Find::MAX_PROBES (duplicated rather than required, to avoid
104
+ # search_backends.rb depending on find.rb) since this backend can be
105
+ # constructed and searched outside Find too.
106
+ TOTAL_CAP = 12
107
+
74
108
  def initialize(path: PATH, env: ENV.fetch("PORTAGE_STORES", nil))
75
109
  @path = path
76
110
  @env = env
@@ -80,23 +114,314 @@ module Portage
80
114
 
81
115
  def available? = !entries.empty?
82
116
 
83
- def search(_query, limit: 10) = entries.first(limit)
117
+ # @return [Array<String>] URLs, routed by #categorized_search.
118
+ def search(query, limit: 10) = categorized_search(query, [limit, TOTAL_CAP].min).map { |e| e[:url] }
119
+
120
+ # @return [Array<Hash>] every entry (url:, categories:), untouched
121
+ # by any query — Index::Sources::StoresFile reuses this to seed
122
+ # the local index from the same file, rather than re-parsing
123
+ # stores.yml itself.
124
+ def stores = entries
84
125
 
85
126
  private
86
127
 
128
+ # Named entries always come first (an explicit "buy from <store>"
129
+ # outranks a category guess), then up to PER_CATEGORY_CAP tagged
130
+ # entries per matching category not already named. When no tagged
131
+ # entry matches any of the query's categories at all, this falls
132
+ # back to named entries plus every untagged entry — the pre-Phase-
133
+ # 2a behaviour for an all-untagged stores.yml — but never to every
134
+ # entry once a tagged match exists: a tagged-but-unrelated store
135
+ # stays excluded, which is the crowding this routing exists to fix.
136
+ def categorized_search(query, limit)
137
+ named = named_entries(query)
138
+ category_ids = Classifier.categories_for(query)
139
+ return fallback(named, limit) unless any_tagged_match?(category_ids)
140
+
141
+ remaining = [limit - named.length, 0].max
142
+ matched = by_category(category_ids, remaining, named)
143
+ (named + matched).uniq { |e| e[:url] }.first(limit)
144
+ end
145
+
146
+ def any_tagged_match?(category_ids)
147
+ category_ids.any? { |category_id| entries.any? { |e| e[:categories].include?(category_id) } }
148
+ end
149
+
150
+ # Named entries plus every untagged entry, capped — not every
151
+ # entry: a tagged store whose categories didn't match the query
152
+ # stays out, the same as it would once a tagged match exists.
153
+ def fallback(named, limit)
154
+ (named + entries.select { |e| e[:categories].empty? }).uniq { |e| e[:url] }.first(limit)
155
+ end
156
+
157
+ def by_category(category_ids, limit, exclude)
158
+ picked = []
159
+ category_ids.each do |category_id|
160
+ entries_for_category(category_id, exclude).each do |entry|
161
+ break if picked.length >= limit
162
+
163
+ picked << entry unless picked.include?(entry)
164
+ end
165
+ end
166
+ picked
167
+ end
168
+
169
+ def entries_for_category(category_id, exclude)
170
+ entries.select { |e| e[:categories].include?(category_id) && !exclude.include?(e) }.first(PER_CATEGORY_CAP)
171
+ end
172
+
173
+ # host or bare name ("shop" out of "shop.example") mentioned in the
174
+ # query — the escape hatch for "buy from <store>" regardless of
175
+ # what it's tagged, or for an untagged store the caller named.
176
+ def named_entries(query)
177
+ entries.select { |e| named?(e, query) }
178
+ end
179
+
180
+ def named?(entry, query)
181
+ host = host_of(entry[:url])
182
+ return false if host.empty?
183
+
184
+ label = host.sub(/\Awww\./, "").split(".").first.to_s
185
+ downcased = query.to_s.downcase
186
+ downcased.include?(host) || (label.length > 2 && downcased.include?(label))
187
+ end
188
+
189
+ def host_of(url)
190
+ URI.parse(url).host.to_s.downcase
191
+ rescue URI::InvalidURIError
192
+ ""
193
+ end
194
+
87
195
  def entries
88
- @entries ||= (env_entries + file_entries).map { |e| e.to_s.strip }.reject(&:empty?).uniq
196
+ @entries ||= (env_entries + file_entries).uniq { |e| e[:url] }
89
197
  end
90
198
 
91
- def env_entries = @env.to_s.split(",")
199
+ def env_entries
200
+ @env.to_s.split(",").map(&:strip).reject(&:empty?).map { |url| { url: url, categories: [] } }
201
+ end
92
202
 
93
203
  def file_entries
94
204
  return [] unless File.readable?(@path)
95
205
 
96
- Array(YAML.safe_load_file(@path))
206
+ Array(YAML.safe_load_file(@path)).filter_map { |entry| normalize(entry) }
97
207
  rescue StandardError
98
208
  []
99
209
  end
210
+
211
+ # A bare string (the format before Phase 2a, still the common case)
212
+ # or a `{url:, categories: [...]}` hash — both parse, per the plan's
213
+ # "the file stays a bare URL list; an entry becomes tagged only when
214
+ # it carries categories:".
215
+ def normalize(entry)
216
+ case entry
217
+ when String
218
+ url = entry.strip
219
+ url.empty? ? nil : { url: url, categories: [] }
220
+ when Hash
221
+ normalize_hash(entry)
222
+ end
223
+ end
224
+
225
+ def normalize_hash(entry)
226
+ entry = entry.transform_keys(&:to_s)
227
+ url = entry["url"].to_s.strip
228
+ return nil if url.empty?
229
+
230
+ { url: url, categories: Array(entry["categories"]).map(&:to_s) }
231
+ end
232
+ end
233
+
234
+ # `~/.portage/index/stores.json` — origins `portage index build`
235
+ # found and verified itself (docs/plans/buy-skill-and-local-browser.md
236
+ # Phase 2b). Unlike Allowlist, this data is **untrusted**: nothing
237
+ # here was ever typed in by the user, so an entry never skips a probe
238
+ # (Find still re-verifies it through ProbeCache like any other
239
+ # candidate URL) and never becomes a `merchant_allowlist`/`--yes`
240
+ # shortcut — it's just another URL a search backend handed back,
241
+ # ranked below Allowlist and above the web-search backends in
242
+ # SearchBackends.default (search_backends_spec.rb has specs proving
243
+ # both non-shortcuts).
244
+ #
245
+ # Routes the same way Allowlist routes a tagged stores.yml (up to
246
+ # PER_CATEGORY_CAP per matching category, TOTAL_CAP overall), plus
247
+ # one thing stores.yml can't do: match a query against a *product* the
248
+ # index has seen (by name or GTIN) and put that product's own stores
249
+ # first, ahead of a category guess. A store the query names outright
250
+ # comes next (Phase 3 — the only route an uncategorised
251
+ # `portage browser import` domain ever gets).
252
+ #
253
+ # Phase 2c: also draws on the repo's own known-stores cache
254
+ # (Index::KnownCache) — fetched lazily the first time this backend is
255
+ # asked for anything and no cache exists yet — merged *underneath*
256
+ # the user's own Store/ProductStore entries: a known entry only shows
257
+ # up when the user's own index doesn't already have that origin/key,
258
+ # so a local `index add`/`index build` finding always wins. Still the
259
+ # same untrusted posture either way — an offer built from either
260
+ # source carries `source: "index"`, never a merchant_allowlist/--yes
261
+ # shortcut (see this class's own header comment).
262
+ class Index
263
+ PER_CATEGORY_CAP = 3
264
+ TOTAL_CAP = 12
265
+
266
+ def initialize(stores: Portage::Cli::Index::Store.new, products: Portage::Cli::Index::ProductStore.new,
267
+ known: Portage::Cli::Index::KnownCache.new)
268
+ @stores = stores
269
+ @products = products
270
+ @known = known
271
+ end
272
+
273
+ def name = "index"
274
+
275
+ def available? = !store_entries.empty? || !product_entries.empty?
276
+
277
+ def search(query, limit: 10)
278
+ return [] if store_entries.empty? && product_entries.empty?
279
+
280
+ product_origins = origins_for_products(query)
281
+ named = named_origins(query, exclude: product_origins)
282
+ category_origins = origins_for_categories(Classifier.categories_for(query), exclude: product_origins + named)
283
+ (product_origins + named + category_origins).uniq.first([limit, TOTAL_CAP].min)
284
+ end
285
+
286
+ private
287
+
288
+ def store_entries
289
+ @store_entries ||= merge_under_own(@stores.all, known_stores, key: "origin")
290
+ end
291
+
292
+ def product_entries
293
+ @product_entries ||= merge_under_own(@products.all, known_products, key: "key")
294
+ end
295
+
296
+ # `entry[key]` is always present on both sides — Store#upsert seeds
297
+ # "origin" and ProductStore#upsert seeds "key" the same way on a
298
+ # brand-new entry, and KnownCache's own fetch just carries whatever
299
+ # `portage index build --export` wrote, in the same schema.
300
+ def merge_under_own(own, known, key:)
301
+ own_keys = own.map { |e| e[key] }
302
+ own + known.reject { |e| own_keys.include?(e[key]) }
303
+ end
304
+
305
+ def known_stores
306
+ ensure_known_cache!
307
+ @known.stores.values
308
+ end
309
+
310
+ def known_products
311
+ ensure_known_cache!
312
+ @known.products.values
313
+ end
314
+
315
+ # Fetched at most once per instance — a cache miss found here means
316
+ # "still no cache", not "try again this call".
317
+ def ensure_known_cache!
318
+ return if @known_cache_checked
319
+
320
+ @known_cache_checked = true
321
+ @known.fetch_if_missing!
322
+ end
323
+
324
+ def origins_for_products(query)
325
+ matches = product_entries.select { |product| product_matches?(product, query) }
326
+ matches.flat_map { |product| Array(product["stores"]).map { |s| s["origin"] } }.uniq
327
+ end
328
+
329
+ # Whole-word matching, built on the same Classifier.tokenize/
330
+ # .word_match? a title/query is classified into categories with
331
+ # (docs/plans/buy-skill-and-local-browser.md Phase 2a) — a plain
332
+ # substring check goes both ways regardless of word boundaries
333
+ # ("tea" inside "steam"/"teak", "bag" inside "bagel"), which is
334
+ # exactly the false-positive class 2a already fixed for category
335
+ # keywords and this backend was still exposed to.
336
+ #
337
+ # A match is either every one of the query's tokens found among the
338
+ # title/alias's own tokens (a short query naming a longer title,
339
+ # e.g. "hiking boots" -> "Men's Hiking Boot"), or the reverse (a
340
+ # longer query that names the whole title/alias as a phrase, e.g.
341
+ # "where can I buy a trail boot"). An empty/whitespace-only query
342
+ # tokenizes to nothing and matches no product.
343
+ def product_matches?(product, query)
344
+ return true if gtin_match?(product, query)
345
+
346
+ # Checked after GTIN, not before: a purely numeric query (the
347
+ # normal shape of a GTIN) tokenizes to nothing at all —
348
+ # Classifier.tokenize splits on runs of non-alpha characters — so
349
+ # gating on "any tokens" first would refuse a valid barcode
350
+ # lookup before #gtin_match? ever got to compare it.
351
+ query_tokens = Classifier.tokenize(query)
352
+ return false if query_tokens.empty?
353
+
354
+ [product["title"], *Array(product["aliases"])].compact.any? { |text| title_matches?(query_tokens, text) }
355
+ end
356
+
357
+ # Exact match against the whole (stripped, downcased) query, never a
358
+ # substring — a GTIN is a barcode, not a word Classifier.tokenize
359
+ # would even keep (it splits on runs of non-alpha characters, so a
360
+ # purely numeric query tokenizes to nothing).
361
+ def gtin_match?(product, query)
362
+ gtin = product["gtin"]
363
+ !gtin.to_s.empty? && query.to_s.strip.downcase == gtin.to_s.downcase
364
+ end
365
+
366
+ def title_matches?(query_tokens, text)
367
+ title_tokens = Classifier.tokenize(text)
368
+ return false if title_tokens.empty?
369
+
370
+ all_match?(query_tokens, title_tokens) || all_match?(title_tokens, query_tokens)
371
+ end
372
+
373
+ def all_match?(these, those)
374
+ these.all? { |a| those.any? { |b| Classifier.word_match?(a, b) } }
375
+ end
376
+
377
+ # A store the query names outright — its host ("allbirds.com") or
378
+ # its bare name as a whole word ("buy from allbirds") — whether or
379
+ # not it carries categories. This is the only way an uncategorised
380
+ # entry (a `portage browser import` domain whose titles matched no
381
+ # category — Phase 3) is ever routed: by name, never for a generic
382
+ # query.
383
+ def named_origins(query, exclude:)
384
+ tokens = Classifier.tokenize(query)
385
+ text = query.to_s.downcase
386
+ store_entries.filter_map do |entry|
387
+ origin = entry["origin"]
388
+ origin if !exclude.include?(origin) && names_store?(origin, text, tokens)
389
+ end
390
+ end
391
+
392
+ def names_store?(origin, text, tokens)
393
+ host = URI.parse(origin.to_s).host.to_s.downcase.delete_prefix("www.")
394
+ return false if host.empty?
395
+
396
+ text.include?(host) || tokens.include?(host.split(".").first)
397
+ rescue URI::InvalidURIError
398
+ false
399
+ end
400
+
401
+ def origins_for_categories(category_ids, exclude:)
402
+ picked = []
403
+ category_ids.each do |category_id|
404
+ entries_for_category(category_id, exclude + picked).each do |origin|
405
+ break if picked.length >= TOTAL_CAP
406
+
407
+ picked << origin
408
+ end
409
+ end
410
+ picked
411
+ end
412
+
413
+ # Ranked by this category's own weight (how many of the store's
414
+ # products the Classifier put there — see Index::Builder's
415
+ # #merge_categories), highest first, before PER_CATEGORY_CAP cuts
416
+ # it off, so a store barely tagged into a category doesn't take a
417
+ # slot from one the Classifier weighted heavily into it.
418
+ def entries_for_category(category_id, exclude)
419
+ matches = store_entries.select do |e|
420
+ Array(e["categories"]&.keys).include?(category_id) && !exclude.include?(e["origin"])
421
+ end
422
+ ranked = matches.sort_by { |e| -e["categories"][category_id].to_i }
423
+ ranked.first(PER_CATEGORY_CAP).map { |e| e["origin"] }
424
+ end
100
425
  end
101
426
 
102
427
  # DuckDuckGo's Instant Answer API — official, documented, no key
@@ -0,0 +1,67 @@
1
+ require "io/console"
2
+
3
+ module Portage
4
+ module Cli
5
+ class SetupWizard
6
+ # The one place every step reads a question and prints a line — so
7
+ # "Enter/blank keeps the current value", "y/N confirm" and "never
8
+ # echo a secret" (docs/plans/buy-skill-and-local-browser.md Phase 4)
9
+ # are each implemented once instead of per step. Defaults to the real
10
+ # $stdin/$stdout, like WebmcpMappingConfirm/BrowserImport::Confirm —
11
+ # a spec swaps in a fake input/output (or stubs the real $stdin the
12
+ # same way cli_spec.rb already does for `browser import`).
13
+ class Prompt
14
+ def initialize(input: $stdin, output: $stdout)
15
+ @input = input
16
+ @output = output
17
+ end
18
+
19
+ def say(text) = @output.puts(text)
20
+
21
+ # @return [Boolean] `default` when the answer is blank; otherwise
22
+ # whether the answer starts with "y".
23
+ def confirm(question, default: true)
24
+ @output.print("#{question} [#{default ? 'Y/n' : 'y/N'}] ")
25
+ @output.flush
26
+ answer = @input.gets.to_s.strip.downcase
27
+ return default if answer.empty?
28
+
29
+ answer.start_with?("y")
30
+ end
31
+
32
+ # @return [String, nil] nil for a blank answer — Enter keeps
33
+ # whatever's already set, and the step decides what that means.
34
+ def ask(question, hint: nil)
35
+ @output.print("#{question}#{" (#{hint})" if hint}: ")
36
+ @output.flush
37
+ blank_to_nil(@input.gets.to_s.chomp)
38
+ end
39
+
40
+ # Same contract as #ask, but the answer is never echoed to the
41
+ # terminal: IO#noecho when `@input` is a real, attached tty (a
42
+ # spec's stubbed-but-otherwise-real $stdin isn't, so this falls
43
+ # back to a plain read there, same as a piped stdin in production).
44
+ def ask_secret(question, hint: nil)
45
+ @output.print("#{question}#{" (#{hint})" if hint}: ")
46
+ @output.flush
47
+ answer = read_secret
48
+ @output.puts
49
+ blank_to_nil(answer.to_s.chomp)
50
+ end
51
+
52
+ private
53
+
54
+ def blank_to_nil(text)
55
+ text = text.strip
56
+ text.empty? ? nil : text
57
+ end
58
+
59
+ def read_secret
60
+ @input.noecho(&:gets).to_s
61
+ rescue Errno::ENOTTY, NoMethodError, IOError
62
+ @input.gets.to_s
63
+ end
64
+ end
65
+ end
66
+ end
67
+ end