portage-cli 0.9.0 → 0.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (34) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +64 -0
  3. data/README.md +33 -4
  4. data/known-stores/categories.yml +6841 -18
  5. data/known-stores/category-stoplist.yml +40 -0
  6. data/known-stores/category-synonyms.yml +15 -0
  7. data/lib/portage/cli/browser_import/categorize.rb +5 -2
  8. data/lib/portage/cli/buy.rb +3 -24
  9. data/lib/portage/cli/check.rb +164 -0
  10. data/lib/portage/cli/check_next_step.rb +51 -0
  11. data/lib/portage/cli/classifier/ranking.rb +135 -0
  12. data/lib/portage/cli/classifier/table.rb +63 -0
  13. data/lib/portage/cli/classifier.rb +23 -44
  14. data/lib/portage/cli/doctor.rb +16 -6
  15. data/lib/portage/cli/find.rb +5 -3
  16. data/lib/portage/cli/handoff_host.rb +33 -0
  17. data/lib/portage/cli/index/builder.rb +49 -12
  18. data/lib/portage/cli/index/database.rb +150 -0
  19. data/lib/portage/cli/index/entry_product.rb +37 -0
  20. data/lib/portage/cli/index/legacy_import.rb +54 -0
  21. data/lib/portage/cli/index/product_store.rb +73 -35
  22. data/lib/portage/cli/index/schema.rb +70 -0
  23. data/lib/portage/cli/index/search.rb +73 -0
  24. data/lib/portage/cli/index/sources/storefront_products/mapper.rb +127 -0
  25. data/lib/portage/cli/index/sources/storefront_products/pages.rb +114 -0
  26. data/lib/portage/cli/index/sources/storefront_products/robots.rb +70 -0
  27. data/lib/portage/cli/index/sources/storefront_products.rb +147 -0
  28. data/lib/portage/cli/index/sources.rb +5 -2
  29. data/lib/portage/cli/index/store.rb +23 -47
  30. data/lib/portage/cli/index.rb +1 -0
  31. data/lib/portage/cli/offer_sources.rb +17 -2
  32. data/lib/portage/cli/version.rb +1 -1
  33. data/lib/portage/cli.rb +142 -12
  34. metadata +34 -4
@@ -0,0 +1,73 @@
1
+ require "uri"
2
+
3
+ module Portage
4
+ module Cli
5
+ module Index
6
+ # The SQL behind ProductStore#search: an FTS5 MATCH over products_fts
7
+ # (Index::Schema) ranked by bm25, or the same filters as a LIKE scan
8
+ # when the database has no FTS table.
9
+ module Search
10
+ # bm25 weights for products_fts's columns: key (unindexed), title,
11
+ # brand, category, aliases. A title hit outranks a brand-only one.
12
+ BM25 = "bm25(products_fts, 0.0, 10.0, 3.0, 1.0, 5.0)".freeze
13
+ LIKE_COLUMNS = %w[$.title $.brand $.category $.aliases].freeze
14
+ HOST_LIKE = Array.new(4) { "s.origin LIKE ?" }.join(" OR ").freeze
15
+
16
+ module_function
17
+
18
+ # Letters and digits only, so nothing the user types is FTS5 syntax.
19
+ # A trailing plural ending is dropped and each word matched as a
20
+ # prefix, so "lights" finds "Light" and "Lighting".
21
+ def words(query)
22
+ query.to_s.downcase.scan(/\p{Alnum}+/).map { |w| w.length > 3 ? w.sub(/(?:ies|es|s)\z/, "") : w }
23
+ end
24
+
25
+ def host_of(store)
26
+ return nil if store.to_s.strip.empty?
27
+
28
+ text = store.to_s.strip
29
+ (text.include?("://") ? URI.parse(text).host : text.split(%r{[/?]}).first).to_s.downcase
30
+ rescue URI::InvalidURIError
31
+ text.downcase
32
+ end
33
+
34
+ # @return [Array(String, Array)] sql and binds.
35
+ def sql(words, category:, host:, limit:, fts:)
36
+ match_sql, binds = fts ? fts_match(words) : like_match(words)
37
+ filters = [match_sql]
38
+ if category
39
+ filters << "json_extract(p.data, '$.category') = ?"
40
+ binds << category.to_s
41
+ end
42
+ if host
43
+ filters << "EXISTS (SELECT 1 FROM product_stores s WHERE s.key = p.key AND (#{HOST_LIKE}))"
44
+ binds.concat(host_patterns(host))
45
+ end
46
+ [select(fts, filters.join(" AND ")), binds + [limit.to_i]]
47
+ end
48
+
49
+ # The host itself or any subdomain of it, with or without a port —
50
+ # so `--store jbhifi.com.au` finds https://www.jbhifi.com.au.
51
+ def host_patterns(host) = ["%://#{host}", "%://#{host}:%", "%.#{host}", "%.#{host}:%"]
52
+
53
+ def select(fts, where)
54
+ if fts
55
+ "SELECT p.data FROM products_fts JOIN products p ON p.id = products_fts.rowid " \
56
+ "WHERE #{where} ORDER BY #{BM25} LIMIT ?"
57
+ else
58
+ "SELECT p.data FROM products p WHERE #{where} ORDER BY p.id LIMIT ?"
59
+ end
60
+ end
61
+
62
+ def fts_match(words)
63
+ ["products_fts MATCH ?", [words.map { |w| %("#{w}"*) }.join(" ")]]
64
+ end
65
+
66
+ def like_match(words)
67
+ any_column = LIKE_COLUMNS.map { |path| "lower(json_extract(p.data, '#{path}')) LIKE ?" }.join(" OR ")
68
+ [words.map { "(#{any_column})" }.join(" AND "), words.flat_map { |w| ["%#{w}%"] * LIKE_COLUMNS.length }]
69
+ end
70
+ end
71
+ end
72
+ end
73
+ end
@@ -0,0 +1,127 @@
1
+ require "portage/ucp"
2
+
3
+ require_relative "../../../classifier"
4
+
5
+ module Portage
6
+ module Cli
7
+ module Index
8
+ module Sources
9
+ class StorefrontProducts
10
+ # One Shopify `/products.json` product -> Portage::Ucp::Product
11
+ # (docs/plans/local-catalogue.md Phase 2), then the index sighting
12
+ # taken from that Product. The product shape is UCP's, never a
13
+ # parallel "card" vocabulary.
14
+ #
15
+ # Not portage-ucp-shopify's Mapper: that one reads Storefront
16
+ # GraphQL nodes (camelCase, gids, MoneyV2 with a currency), not the
17
+ # REST products.json shape, and portage-cli doesn't depend on that
18
+ # gem.
19
+ #
20
+ # The Product carries what the store sent, price and availability
21
+ # included (products.json has no currency, so Price#currency is
22
+ # nil). #sighting is what the index persists, and it takes only
23
+ # identity fields from the Product: price and availability never
24
+ # reach it.
25
+ module Mapper
26
+ MAX_CATEGORIES = 3
27
+ TAXONOMY = "google_product_category".freeze
28
+ # Shopify's own placeholder for a product with no real options.
29
+ PLACEHOLDER_OPTION = { "name" => "Title", "values" => ["Default Title"] }.freeze
30
+
31
+ module_function
32
+
33
+ # @param raw [Hash] one entry of products.json's "products".
34
+ # @param classify [#call] text -> category ids (Classifier).
35
+ def product(raw, origin:, classify: Classifier.method(:categories_for))
36
+ options = product_options(raw)
37
+ Portage::Ucp::Product.new(
38
+ id: "gid://shopify/Product/#{raw['id']}", title: raw["title"].to_s,
39
+ description: Portage::Ucp::Description.new(html: raw["body_html"]),
40
+ price_range: price_range(raw), variants: Array(raw["variants"]).map { |v| variant(v, options) },
41
+ handle: raw["handle"], url: "#{origin}/products/#{raw['handle']}",
42
+ categories: categories(raw, classify), media: Array(raw["images"]).map { |i| media(i) },
43
+ options: options.map { |o| option(o) }, tags: Array(raw["tags"])
44
+ )
45
+ end
46
+
47
+ # @return [Hash] an Index::Builder sighting: the ProductStore
48
+ # fields plus `product:`, the extra entry fields this source
49
+ # adds (handle, url, image_url, options, variant_ids).
50
+ def sighting(raw, origin:, classify: Classifier.method(:categories_for))
51
+ product = product(raw, origin: origin, classify: classify)
52
+ { origin: origin, url: product.url, title: product.title, brand: brand(raw), gtin: nil,
53
+ categories: product.categories.select { |c| c.taxonomy == TAXONOMY }.map(&:value),
54
+ product: { handle: product.handle, url: product.url, image_url: product.media.first&.url,
55
+ options: product.options.map(&:to_wire_h), variant_ids: product.variants.map(&:id) } }
56
+ end
57
+
58
+ def brand(raw)
59
+ vendor = raw["vendor"].to_s.strip
60
+ vendor.empty? ? nil : vendor
61
+ end
62
+
63
+ def product_options(raw)
64
+ Array(raw["options"]).reject { |o| o.slice("name", "values") == PLACEHOLDER_OPTION }
65
+ end
66
+
67
+ def option(raw)
68
+ values = Array(raw["values"]).map { |label| Portage::Ucp::OptionValue.new(label: label.to_s) }
69
+ Portage::Ucp::ProductOption.new(name: raw["name"].to_s, values: values)
70
+ end
71
+
72
+ def variant(raw, options)
73
+ Portage::Ucp::Variant.new(
74
+ id: "gid://shopify/ProductVariant/#{raw['id']}", title: raw["title"].to_s,
75
+ description: Portage::Ucp::Description.new(plain: raw["title"].to_s), price: price(raw["price"]),
76
+ sku: raw["sku"].to_s.empty? ? nil : raw["sku"],
77
+ list_price: raw["compare_at_price"] ? price(raw["compare_at_price"]) : nil,
78
+ availability: raw.key?("available") ? { "available" => raw["available"] } : nil,
79
+ options: selected_options(raw, options), media: variant_media(raw)
80
+ )
81
+ end
82
+
83
+ # option1..option3 line up with the product's options by
84
+ # position; the placeholder option is already gone, so a
85
+ # "Default Title" variant selects nothing.
86
+ def selected_options(raw, options)
87
+ options.each_with_index.filter_map do |opt, i|
88
+ label = raw["option#{i + 1}"]
89
+ Portage::Ucp::SelectedOption.new(name: opt["name"].to_s, label: label.to_s) if label
90
+ end
91
+ end
92
+
93
+ def variant_media(raw) = raw["featured_image"] ? [media(raw["featured_image"])] : []
94
+
95
+ def media(raw)
96
+ Portage::Ucp::Media.new(type: "image", url: raw["src"], alt_text: raw["alt"], width: raw["width"],
97
+ height: raw["height"])
98
+ end
99
+
100
+ def price(amount)
101
+ Portage::Ucp::Price.new(amount: Portage::Ucp::Support::Amounts.decimal_to_minor(amount), currency: nil)
102
+ end
103
+
104
+ def price_range(raw)
105
+ amounts = Array(raw["variants"]).map { |v| v["price"] }.compact
106
+ amounts = ["0"] if amounts.empty?
107
+ minors = amounts.map { |a| Portage::Ucp::Support::Amounts.decimal_to_minor(a) }
108
+ Portage::Ucp::PriceRange.new(min: Portage::Ucp::Price.new(amount: minors.min, currency: nil),
109
+ max: Portage::Ucp::Price.new(amount: minors.max, currency: nil))
110
+ end
111
+
112
+ # Classifier on product_type and tags (the plan's inputs), top
113
+ # three ids, then the store's own product_type as a merchant
114
+ # category.
115
+ def categories(raw, classify)
116
+ text = [raw["product_type"], *Array(raw["tags"])].compact.join(" ")
117
+ ids = text.strip.empty? ? [] : classify.call(text).first(MAX_CATEGORIES)
118
+ google = ids.map { |id| Portage::Ucp::Category.new(value: id, taxonomy: TAXONOMY) }
119
+ type = raw["product_type"].to_s.strip
120
+ type.empty? ? google : google + [Portage::Ucp::Category.new(value: type, taxonomy: "merchant")]
121
+ end
122
+ end
123
+ end
124
+ end
125
+ end
126
+ end
127
+ end
@@ -0,0 +1,114 @@
1
+ require "net/http"
2
+ require "uri"
3
+ require "json"
4
+ require "portage/ucp/support/connection"
5
+
6
+ require_relative "../../../user_agent"
7
+
8
+ module Portage
9
+ module Cli
10
+ module Index
11
+ module Sources
12
+ class StorefrontProducts
13
+ # The HTTP half of a crawl: GETs `products.json?limit=&page=`
14
+ # through Support::Connection with the shared UserAgent, pausing
15
+ # between pages, waiting out one 429's Retry-After and stopping on
16
+ # a second, and turning anything that isn't a products array into
17
+ # a reason.
18
+ class Pages
19
+ PAUSE = 1
20
+ MAX_RETRY_AFTER = 60
21
+ DEFAULT_RETRY_AFTER = 10
22
+ OPEN_TIMEOUT = 5
23
+ READ_TIMEOUT = 20
24
+
25
+ def self.get(url, accept: "application/json")
26
+ uri = URI.parse(url)
27
+ headers = UserAgent.headers.merge("Accept" => accept)
28
+ Portage::Ucp::Support::Connection.start(uri, route: :store, open_timeout: OPEN_TIMEOUT,
29
+ read_timeout: READ_TIMEOUT) do |http|
30
+ http.get(uri.request_uri, headers)
31
+ end
32
+ end
33
+
34
+ # @param allowed [#call] path -> whether robots.txt allows it.
35
+ def initialize(endpoint, per_page:, max_pages:, sleeper:, allowed: ->(_path) { true })
36
+ @endpoint = endpoint
37
+ @path = URI.parse(endpoint).path
38
+ @allowed = allowed
39
+ @per_page = per_page
40
+ @max_pages = max_pages
41
+ @sleeper = sleeper
42
+ @rate_limited = 0
43
+ end
44
+
45
+ # Yields each non-empty page's raw products.
46
+ # @return [Array(String, String, Integer)] status ("ok",
47
+ # "partial", "skipped"), reason (nil when ok) and pages read.
48
+ def each(&)
49
+ (1..@max_pages).each do |page|
50
+ @sleeper.call(PAUSE) if page > 1
51
+ products = fetch(page)
52
+ done = outcome(products, page, &)
53
+ return done if done
54
+ end
55
+ ["partial", "page_cap", @max_pages]
56
+ end
57
+
58
+ private
59
+
60
+ # nil means "full page, keep going".
61
+ def outcome(products, page)
62
+ return failed(products, page - 1) unless products.is_a?(Array)
63
+ return ["skipped", "empty", 0] if products.empty? && page == 1
64
+
65
+ yield products unless products.empty?
66
+ ["ok", nil, page] if products.length < @per_page
67
+ end
68
+
69
+ # Past page 1 a failure keeps what earlier pages found.
70
+ def failed(reason, pages) = [pages.zero? ? "skipped" : "partial", reason, pages]
71
+
72
+ # @return [Array<Hash>, String] a page's products, or why there
73
+ # are none.
74
+ def fetch(page)
75
+ return "robots" unless @allowed.call("#{@path}#{query(page)}")
76
+
77
+ response = self.class.get("#{@endpoint}#{query(page)}")
78
+ return products_in(response) unless response.code == "429"
79
+
80
+ @rate_limited += 1
81
+ return "rate_limited" if @rate_limited > 1
82
+
83
+ @sleeper.call(retry_after(response["Retry-After"]))
84
+ fetch(page)
85
+ rescue StandardError => e
86
+ "error: #{e.class}"
87
+ end
88
+
89
+ # Page 1 is the bare URL (Shopify's default page), so a robots
90
+ # rule aimed at duplicate `?page=1` URLs doesn't read as a ban
91
+ # on the whole catalogue.
92
+ def query(page) = page == 1 ? "?limit=#{@per_page}" : "?limit=#{@per_page}&page=#{page}"
93
+
94
+ def products_in(response)
95
+ return "not_found" if response.code == "404"
96
+ return "redirect" if response.is_a?(Net::HTTPRedirection)
97
+ return "http_#{response.code}" unless response.is_a?(Net::HTTPSuccess)
98
+
99
+ body = JSON.parse(response.body.to_s.dup.force_encoding(Encoding::UTF_8))
100
+ body.is_a?(Hash) && body["products"].is_a?(Array) ? body["products"] : "not_json"
101
+ rescue JSON::ParserError
102
+ "not_json"
103
+ end
104
+
105
+ def retry_after(value)
106
+ seconds = Integer(value.to_s, exception: false) || DEFAULT_RETRY_AFTER
107
+ seconds.clamp(0, MAX_RETRY_AFTER)
108
+ end
109
+ end
110
+ end
111
+ end
112
+ end
113
+ end
114
+ end
@@ -0,0 +1,70 @@
1
+ module Portage
2
+ module Cli
3
+ module Index
4
+ module Sources
5
+ class StorefrontProducts
6
+ # Just enough robots.txt (RFC 9309) to answer "may this agent GET
7
+ # this path?": the group naming this agent if there is one, else
8
+ # the `*` group; `*` and `$` in rules; the longest matching rule
9
+ # wins, and Allow wins a tie. No rules, or no file, means allowed.
10
+ class Robots
11
+ def initialize(body)
12
+ @groups = parse(body.to_s)
13
+ end
14
+
15
+ # @param path [String] request path, query included.
16
+ # @param agent [String] the User-Agent this process sends.
17
+ def allowed?(path, agent:)
18
+ rules = rules_for(agent.to_s.downcase)
19
+ matches = rules.select { |_allow, pattern| match?(pattern, path) }
20
+ return true if matches.empty?
21
+
22
+ longest = matches.map { |_allow, pattern| pattern.length }.max
23
+ matches.select { |_allow, pattern| pattern.length == longest }.any?(&:first)
24
+ end
25
+
26
+ private
27
+
28
+ def rules_for(agent)
29
+ named = @groups.select { |agents, *| agents.any? { |a| a != "*" && agent.include?(a) } }
30
+ chosen = named.empty? ? @groups.select { |agents, *| agents.include?("*") } : named
31
+ chosen.flat_map { |_agents, rules, _closed| rules }
32
+ end
33
+
34
+ # @return [Array<Array>] [agents, [[allow?, pattern], ...], closed]
35
+ # per group.
36
+ def parse(body)
37
+ groups = []
38
+ body.each_line do |line|
39
+ field, value = line.sub(/#.*/, "").split(":", 2).map { |part| part.to_s.strip }
40
+ next if value.nil?
41
+
42
+ add_line(groups, field.downcase, value)
43
+ end
44
+ groups
45
+ end
46
+
47
+ # A rule line (even an empty `Disallow:`) ends a group's
48
+ # User-agent list, so the next User-agent starts a new group.
49
+ def add_line(groups, field, value)
50
+ if field == "user-agent"
51
+ groups << [[], [], false] if groups.empty? || groups.last[2]
52
+ groups.last[0] << value.downcase
53
+ elsif %w[allow disallow].include?(field) && groups.any?
54
+ groups.last[2] = true
55
+ groups.last[1] << [field == "allow", value] unless value.empty?
56
+ end
57
+ end
58
+
59
+ def match?(pattern, path)
60
+ anchored = pattern.end_with?("$")
61
+ body = anchored ? pattern.chomp("$") : pattern
62
+ regex = body.split("*", -1).map { |part| Regexp.escape(part) }.join(".*")
63
+ Regexp.new("\\A#{regex}#{'\\z' if anchored}").match?(path)
64
+ end
65
+ end
66
+ end
67
+ end
68
+ end
69
+ end
70
+ end
@@ -0,0 +1,147 @@
1
+ require "net/http"
2
+ require "uri"
3
+
4
+ require_relative "../../user_agent"
5
+ require_relative "../../classifier"
6
+ require_relative "../../handoff_only"
7
+ require_relative "../store"
8
+ require_relative "storefront_products/mapper"
9
+ require_relative "storefront_products/robots"
10
+ require_relative "storefront_products/pages"
11
+
12
+ module Portage
13
+ module Cli
14
+ module Index
15
+ module Sources
16
+ # A store's own public catalogue, read from Shopify's `/products.json`
17
+ # (docs/plans/local-catalogue.md Phase 2), for origins already in
18
+ # the index. Off by default: it runs only from `portage index build
19
+ # --sources storefront_products` or `portage index add URL --crawl`,
20
+ # never from `find` or `buy`.
21
+ #
22
+ # Guardrails: a hand-off-only host (HandoffOnly, checked live) and a
23
+ # robots.txt Disallow get no products.json request at all; pages are
24
+ # MAX_PAGES x 250 at most, PAUSE seconds apart; a 429 waits out its
25
+ # Retry-After once and a second 429 stops that store; a 404, a
26
+ # redirect, an empty page 1, or anything that isn't products JSON
27
+ # (a bot wall) skips the store. Every outcome is noted on the store
28
+ # row as `crawl` ({status, reason, pages, products, at}).
29
+ #
30
+ # Yields Mapper sightings (no price, no availability) plus one store
31
+ # sighting per origin carrying `store_fields:` (the crawl note, and
32
+ # platform "shopify" when products.json answered); Index::Builder
33
+ # writes both.
34
+ class StorefrontProducts
35
+ MAX_PAGES = 20
36
+ MAX_STORES = 25
37
+ PER_PAGE = 250
38
+
39
+ # The one platform seam: WooCommerce's Store API can join here.
40
+ # nil platform means "not known yet", worth a try.
41
+ def self.endpoint_for(origin, platform)
42
+ "#{origin}/products.json" if platform.nil? || platform == "shopify"
43
+ end
44
+
45
+ # The guardrails, plus test seams for sleeping and the clock.
46
+ def initialize(stores: Store.new, handoff_only: HandoffOnly.new, max_pages: MAX_PAGES,
47
+ max_stores: MAX_STORES, per_page: PER_PAGE, sleeper: Kernel.method(:sleep), now: Time.now)
48
+ @stores = stores
49
+ @handoff_only = handoff_only
50
+ @max_pages = max_pages
51
+ @max_stores = max_stores
52
+ @per_page = per_page
53
+ @sleeper = sleeper
54
+ @now = now
55
+ end
56
+
57
+ def name = "storefront_products"
58
+
59
+ def description
60
+ "Each indexed store's own /products.json (Shopify), up to #{MAX_PAGES} pages a store and " \
61
+ "#{MAX_STORES} stores a run, least recently crawled first. Off by default — opt in with " \
62
+ "--sources storefront_products or `index add URL --crawl`."
63
+ end
64
+
65
+ def source_path = nil
66
+
67
+ # `**` accepts (and ignores) the shared Source#candidates(queries:)
68
+ # interface.
69
+ def candidates(**)
70
+ eligible = @stores.all.select do |entry|
71
+ self.class.endpoint_for(entry["origin"], entry["platform"]) && !handoff_only?(entry["origin"])
72
+ end
73
+ eligible.sort_by { |entry| entry.dig("crawl", "at").to_i }.first(@max_stores)
74
+ .flat_map { |entry| crawl(entry["origin"], platform: entry["platform"]) }
75
+ end
76
+
77
+ # @return [Array<Hash>] product sightings, then the store sighting.
78
+ def crawl(origin, platform: nil)
79
+ endpoint = self.class.endpoint_for(origin, platform)
80
+ robots = robots_for(origin, endpoint)
81
+ return [store_sighting(origin, note("skipped", robots, 0, 0))] if robots.is_a?(String)
82
+
83
+ sightings = []
84
+ classify = memoized_classifier
85
+ status, reason, pages = pages_for(endpoint, robots).each do |products|
86
+ sightings.concat(products.map { |raw| Mapper.sighting(raw, origin: origin, classify: classify) })
87
+ end
88
+ sightings + [store_sighting(origin, note(status, reason, pages, sightings.length))]
89
+ rescue StandardError => e
90
+ [store_sighting(origin, note("skipped", "error: #{e.class}", 0, 0))]
91
+ end
92
+
93
+ private
94
+
95
+ def pages_for(endpoint, robots)
96
+ Pages.new(endpoint, per_page: @per_page, max_pages: @max_pages, sleeper: @sleeper,
97
+ allowed: ->(path) { robots.allowed?(path, agent: UserAgent.value) })
98
+ end
99
+
100
+ # The store's robots.txt rules, or why the crawl stops before
101
+ # any request (no endpoint, hand-off only) or before any
102
+ # products.json request (robots.txt unreachable). Pages checks
103
+ # the rules against every page URL it would request.
104
+ # RFC 9309: a 4xx robots.txt means no rules; a 5xx or no answer
105
+ # at all means stay out.
106
+ # @return [Robots, String]
107
+ def robots_for(origin, endpoint)
108
+ return "unsupported_platform" unless endpoint
109
+ return "handoff_only" if handoff_only?(origin)
110
+
111
+ response = Pages.get("#{origin}/robots.txt", accept: "text/plain")
112
+ return Robots.new(nil) if response.code.start_with?("4")
113
+ return "redirect" if response.is_a?(Net::HTTPRedirection)
114
+ return "robots_unreachable" unless response.is_a?(Net::HTTPSuccess)
115
+
116
+ Robots.new(response.body.to_s.dup.force_encoding(Encoding::UTF_8))
117
+ rescue StandardError
118
+ "robots_unreachable"
119
+ end
120
+
121
+ # Product types and tags repeat across a catalogue, and Classifier
122
+ # re-reads its YAML on every call.
123
+ def memoized_classifier
124
+ memo = {}
125
+ ->(text) { memo[text] ||= Classifier.categories_for(text) }
126
+ end
127
+
128
+ def handoff_only?(origin)
129
+ @handoff_only.host?(URI.parse(origin.to_s).host)
130
+ rescue URI::InvalidURIError
131
+ true
132
+ end
133
+
134
+ def note(status, reason, pages, products)
135
+ { "status" => status, "reason" => reason, "pages" => pages, "products" => products, "at" => @now.to_i }
136
+ end
137
+
138
+ def store_sighting(origin, crawl)
139
+ fields = { crawl: crawl }
140
+ fields[:platform] = "shopify" unless crawl["status"] == "skipped"
141
+ { origin: origin, url: nil, title: nil, brand: nil, gtin: nil, store_fields: fields }
142
+ end
143
+ end
144
+ end
145
+ end
146
+ end
147
+ end
@@ -3,6 +3,7 @@ require_relative "sources/stores_file"
3
3
  require_relative "sources/browser"
4
4
  require_relative "sources/wikidata"
5
5
  require_relative "sources/webmcp_sweep"
6
+ require_relative "sources/storefront_products"
6
7
 
7
8
  module Portage
8
9
  module Cli
@@ -16,7 +17,8 @@ module Portage
16
17
  "stores_file" => -> { StoresFile.new },
17
18
  "browser" => -> { Browser.new },
18
19
  "wikidata" => -> { Wikidata.new },
19
- "webmcp_sweep" => -> { WebmcpSweep.new }
20
+ "webmcp_sweep" => -> { WebmcpSweep.new },
21
+ "storefront_products" => -> { StorefrontProducts.new }
20
22
  }.freeze
21
23
 
22
24
  # Runs with no extra opt-in: no bridge required, and no low-yield
@@ -24,7 +26,8 @@ module Portage
24
26
  # show up in `portage index sources`, just not run unless named in
25
27
  # `--sources`. `browser` is listed but never a default — it yields
26
28
  # nothing; `portage browser import` writes those entries itself
27
- # (see Sources::Browser).
29
+ # (see Sources::Browser). `storefront_products` sends up to 21
30
+ # requests a store, so it runs only when named.
28
31
  DEFAULT_NAMES = %w[shopify_catalog stores_file].freeze
29
32
 
30
33
  def self.all = ALL.values.map(&:call)