portage-cli 0.9.0 → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +64 -0
- data/README.md +33 -4
- data/known-stores/categories.yml +6841 -18
- data/known-stores/category-stoplist.yml +40 -0
- data/known-stores/category-synonyms.yml +15 -0
- data/lib/portage/cli/browser_import/categorize.rb +5 -2
- data/lib/portage/cli/buy.rb +3 -24
- data/lib/portage/cli/check.rb +164 -0
- data/lib/portage/cli/check_next_step.rb +51 -0
- data/lib/portage/cli/classifier/ranking.rb +135 -0
- data/lib/portage/cli/classifier/table.rb +63 -0
- data/lib/portage/cli/classifier.rb +23 -44
- data/lib/portage/cli/doctor.rb +16 -6
- data/lib/portage/cli/find.rb +5 -3
- data/lib/portage/cli/handoff_host.rb +33 -0
- data/lib/portage/cli/index/builder.rb +49 -12
- data/lib/portage/cli/index/database.rb +150 -0
- data/lib/portage/cli/index/entry_product.rb +37 -0
- data/lib/portage/cli/index/legacy_import.rb +54 -0
- data/lib/portage/cli/index/product_store.rb +73 -35
- data/lib/portage/cli/index/schema.rb +70 -0
- data/lib/portage/cli/index/search.rb +73 -0
- data/lib/portage/cli/index/sources/storefront_products/mapper.rb +127 -0
- data/lib/portage/cli/index/sources/storefront_products/pages.rb +114 -0
- data/lib/portage/cli/index/sources/storefront_products/robots.rb +70 -0
- data/lib/portage/cli/index/sources/storefront_products.rb +147 -0
- data/lib/portage/cli/index/sources.rb +5 -2
- data/lib/portage/cli/index/store.rb +23 -47
- data/lib/portage/cli/index.rb +1 -0
- data/lib/portage/cli/offer_sources.rb +17 -2
- data/lib/portage/cli/version.rb +1 -1
- data/lib/portage/cli.rb +142 -12
- metadata +34 -4
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
require "uri"
|
|
2
|
+
|
|
3
|
+
module Portage
|
|
4
|
+
module Cli
|
|
5
|
+
module Index
|
|
6
|
+
# The SQL behind ProductStore#search: an FTS5 MATCH over products_fts
|
|
7
|
+
# (Index::Schema) ranked by bm25, or the same filters as a LIKE scan
|
|
8
|
+
# when the database has no FTS table.
|
|
9
|
+
module Search
|
|
10
|
+
# bm25 weights for products_fts's columns: key (unindexed), title,
|
|
11
|
+
# brand, category, aliases. A title hit outranks a brand-only one.
|
|
12
|
+
BM25 = "bm25(products_fts, 0.0, 10.0, 3.0, 1.0, 5.0)".freeze
|
|
13
|
+
LIKE_COLUMNS = %w[$.title $.brand $.category $.aliases].freeze
|
|
14
|
+
HOST_LIKE = Array.new(4) { "s.origin LIKE ?" }.join(" OR ").freeze
|
|
15
|
+
|
|
16
|
+
module_function
|
|
17
|
+
|
|
18
|
+
# Letters and digits only, so nothing the user types is FTS5 syntax.
|
|
19
|
+
# A trailing plural ending is dropped and each word matched as a
|
|
20
|
+
# prefix, so "lights" finds "Light" and "Lighting".
|
|
21
|
+
def words(query)
|
|
22
|
+
query.to_s.downcase.scan(/\p{Alnum}+/).map { |w| w.length > 3 ? w.sub(/(?:ies|es|s)\z/, "") : w }
|
|
23
|
+
end
|
|
24
|
+
|
|
25
|
+
def host_of(store)
|
|
26
|
+
return nil if store.to_s.strip.empty?
|
|
27
|
+
|
|
28
|
+
text = store.to_s.strip
|
|
29
|
+
(text.include?("://") ? URI.parse(text).host : text.split(%r{[/?]}).first).to_s.downcase
|
|
30
|
+
rescue URI::InvalidURIError
|
|
31
|
+
text.downcase
|
|
32
|
+
end
|
|
33
|
+
|
|
34
|
+
# @return [Array(String, Array)] sql and binds.
|
|
35
|
+
def sql(words, category:, host:, limit:, fts:)
|
|
36
|
+
match_sql, binds = fts ? fts_match(words) : like_match(words)
|
|
37
|
+
filters = [match_sql]
|
|
38
|
+
if category
|
|
39
|
+
filters << "json_extract(p.data, '$.category') = ?"
|
|
40
|
+
binds << category.to_s
|
|
41
|
+
end
|
|
42
|
+
if host
|
|
43
|
+
filters << "EXISTS (SELECT 1 FROM product_stores s WHERE s.key = p.key AND (#{HOST_LIKE}))"
|
|
44
|
+
binds.concat(host_patterns(host))
|
|
45
|
+
end
|
|
46
|
+
[select(fts, filters.join(" AND ")), binds + [limit.to_i]]
|
|
47
|
+
end
|
|
48
|
+
|
|
49
|
+
# The host itself or any subdomain of it, with or without a port —
|
|
50
|
+
# so `--store jbhifi.com.au` finds https://www.jbhifi.com.au.
|
|
51
|
+
def host_patterns(host) = ["%://#{host}", "%://#{host}:%", "%.#{host}", "%.#{host}:%"]
|
|
52
|
+
|
|
53
|
+
def select(fts, where)
|
|
54
|
+
if fts
|
|
55
|
+
"SELECT p.data FROM products_fts JOIN products p ON p.id = products_fts.rowid " \
|
|
56
|
+
"WHERE #{where} ORDER BY #{BM25} LIMIT ?"
|
|
57
|
+
else
|
|
58
|
+
"SELECT p.data FROM products p WHERE #{where} ORDER BY p.id LIMIT ?"
|
|
59
|
+
end
|
|
60
|
+
end
|
|
61
|
+
|
|
62
|
+
def fts_match(words)
|
|
63
|
+
["products_fts MATCH ?", [words.map { |w| %("#{w}"*) }.join(" ")]]
|
|
64
|
+
end
|
|
65
|
+
|
|
66
|
+
def like_match(words)
|
|
67
|
+
any_column = LIKE_COLUMNS.map { |path| "lower(json_extract(p.data, '#{path}')) LIKE ?" }.join(" OR ")
|
|
68
|
+
[words.map { "(#{any_column})" }.join(" AND "), words.flat_map { |w| ["%#{w}%"] * LIKE_COLUMNS.length }]
|
|
69
|
+
end
|
|
70
|
+
end
|
|
71
|
+
end
|
|
72
|
+
end
|
|
73
|
+
end
|
|
@@ -0,0 +1,127 @@
|
|
|
1
|
+
require "portage/ucp"
|
|
2
|
+
|
|
3
|
+
require_relative "../../../classifier"
|
|
4
|
+
|
|
5
|
+
module Portage
|
|
6
|
+
module Cli
|
|
7
|
+
module Index
|
|
8
|
+
module Sources
|
|
9
|
+
class StorefrontProducts
|
|
10
|
+
# One Shopify `/products.json` product -> Portage::Ucp::Product
|
|
11
|
+
# (docs/plans/local-catalogue.md Phase 2), then the index sighting
|
|
12
|
+
# taken from that Product. The product shape is UCP's, never a
|
|
13
|
+
# parallel "card" vocabulary.
|
|
14
|
+
#
|
|
15
|
+
# Not portage-ucp-shopify's Mapper: that one reads Storefront
|
|
16
|
+
# GraphQL nodes (camelCase, gids, MoneyV2 with a currency), not the
|
|
17
|
+
# REST products.json shape, and portage-cli doesn't depend on that
|
|
18
|
+
# gem.
|
|
19
|
+
#
|
|
20
|
+
# The Product carries what the store sent, price and availability
|
|
21
|
+
# included (products.json has no currency, so Price#currency is
|
|
22
|
+
# nil). #sighting is what the index persists, and it takes only
|
|
23
|
+
# identity fields from the Product: price and availability never
|
|
24
|
+
# reach it.
|
|
25
|
+
module Mapper
|
|
26
|
+
MAX_CATEGORIES = 3
|
|
27
|
+
TAXONOMY = "google_product_category".freeze
|
|
28
|
+
# Shopify's own placeholder for a product with no real options.
|
|
29
|
+
PLACEHOLDER_OPTION = { "name" => "Title", "values" => ["Default Title"] }.freeze
|
|
30
|
+
|
|
31
|
+
module_function
|
|
32
|
+
|
|
33
|
+
# @param raw [Hash] one entry of products.json's "products".
|
|
34
|
+
# @param classify [#call] text -> category ids (Classifier).
|
|
35
|
+
def product(raw, origin:, classify: Classifier.method(:categories_for))
|
|
36
|
+
options = product_options(raw)
|
|
37
|
+
Portage::Ucp::Product.new(
|
|
38
|
+
id: "gid://shopify/Product/#{raw['id']}", title: raw["title"].to_s,
|
|
39
|
+
description: Portage::Ucp::Description.new(html: raw["body_html"]),
|
|
40
|
+
price_range: price_range(raw), variants: Array(raw["variants"]).map { |v| variant(v, options) },
|
|
41
|
+
handle: raw["handle"], url: "#{origin}/products/#{raw['handle']}",
|
|
42
|
+
categories: categories(raw, classify), media: Array(raw["images"]).map { |i| media(i) },
|
|
43
|
+
options: options.map { |o| option(o) }, tags: Array(raw["tags"])
|
|
44
|
+
)
|
|
45
|
+
end
|
|
46
|
+
|
|
47
|
+
# @return [Hash] an Index::Builder sighting: the ProductStore
|
|
48
|
+
# fields plus `product:`, the extra entry fields this source
|
|
49
|
+
# adds (handle, url, image_url, options, variant_ids).
|
|
50
|
+
def sighting(raw, origin:, classify: Classifier.method(:categories_for))
|
|
51
|
+
product = product(raw, origin: origin, classify: classify)
|
|
52
|
+
{ origin: origin, url: product.url, title: product.title, brand: brand(raw), gtin: nil,
|
|
53
|
+
categories: product.categories.select { |c| c.taxonomy == TAXONOMY }.map(&:value),
|
|
54
|
+
product: { handle: product.handle, url: product.url, image_url: product.media.first&.url,
|
|
55
|
+
options: product.options.map(&:to_wire_h), variant_ids: product.variants.map(&:id) } }
|
|
56
|
+
end
|
|
57
|
+
|
|
58
|
+
def brand(raw)
|
|
59
|
+
vendor = raw["vendor"].to_s.strip
|
|
60
|
+
vendor.empty? ? nil : vendor
|
|
61
|
+
end
|
|
62
|
+
|
|
63
|
+
def product_options(raw)
|
|
64
|
+
Array(raw["options"]).reject { |o| o.slice("name", "values") == PLACEHOLDER_OPTION }
|
|
65
|
+
end
|
|
66
|
+
|
|
67
|
+
def option(raw)
|
|
68
|
+
values = Array(raw["values"]).map { |label| Portage::Ucp::OptionValue.new(label: label.to_s) }
|
|
69
|
+
Portage::Ucp::ProductOption.new(name: raw["name"].to_s, values: values)
|
|
70
|
+
end
|
|
71
|
+
|
|
72
|
+
def variant(raw, options)
|
|
73
|
+
Portage::Ucp::Variant.new(
|
|
74
|
+
id: "gid://shopify/ProductVariant/#{raw['id']}", title: raw["title"].to_s,
|
|
75
|
+
description: Portage::Ucp::Description.new(plain: raw["title"].to_s), price: price(raw["price"]),
|
|
76
|
+
sku: raw["sku"].to_s.empty? ? nil : raw["sku"],
|
|
77
|
+
list_price: raw["compare_at_price"] ? price(raw["compare_at_price"]) : nil,
|
|
78
|
+
availability: raw.key?("available") ? { "available" => raw["available"] } : nil,
|
|
79
|
+
options: selected_options(raw, options), media: variant_media(raw)
|
|
80
|
+
)
|
|
81
|
+
end
|
|
82
|
+
|
|
83
|
+
# option1..option3 line up with the product's options by
|
|
84
|
+
# position; the placeholder option is already gone, so a
|
|
85
|
+
# "Default Title" variant selects nothing.
|
|
86
|
+
def selected_options(raw, options)
|
|
87
|
+
options.each_with_index.filter_map do |opt, i|
|
|
88
|
+
label = raw["option#{i + 1}"]
|
|
89
|
+
Portage::Ucp::SelectedOption.new(name: opt["name"].to_s, label: label.to_s) if label
|
|
90
|
+
end
|
|
91
|
+
end
|
|
92
|
+
|
|
93
|
+
def variant_media(raw) = raw["featured_image"] ? [media(raw["featured_image"])] : []
|
|
94
|
+
|
|
95
|
+
def media(raw)
|
|
96
|
+
Portage::Ucp::Media.new(type: "image", url: raw["src"], alt_text: raw["alt"], width: raw["width"],
|
|
97
|
+
height: raw["height"])
|
|
98
|
+
end
|
|
99
|
+
|
|
100
|
+
def price(amount)
|
|
101
|
+
Portage::Ucp::Price.new(amount: Portage::Ucp::Support::Amounts.decimal_to_minor(amount), currency: nil)
|
|
102
|
+
end
|
|
103
|
+
|
|
104
|
+
def price_range(raw)
|
|
105
|
+
amounts = Array(raw["variants"]).map { |v| v["price"] }.compact
|
|
106
|
+
amounts = ["0"] if amounts.empty?
|
|
107
|
+
minors = amounts.map { |a| Portage::Ucp::Support::Amounts.decimal_to_minor(a) }
|
|
108
|
+
Portage::Ucp::PriceRange.new(min: Portage::Ucp::Price.new(amount: minors.min, currency: nil),
|
|
109
|
+
max: Portage::Ucp::Price.new(amount: minors.max, currency: nil))
|
|
110
|
+
end
|
|
111
|
+
|
|
112
|
+
# Classifier on product_type and tags (the plan's inputs), top
|
|
113
|
+
# three ids, then the store's own product_type as a merchant
|
|
114
|
+
# category.
|
|
115
|
+
def categories(raw, classify)
|
|
116
|
+
text = [raw["product_type"], *Array(raw["tags"])].compact.join(" ")
|
|
117
|
+
ids = text.strip.empty? ? [] : classify.call(text).first(MAX_CATEGORIES)
|
|
118
|
+
google = ids.map { |id| Portage::Ucp::Category.new(value: id, taxonomy: TAXONOMY) }
|
|
119
|
+
type = raw["product_type"].to_s.strip
|
|
120
|
+
type.empty? ? google : google + [Portage::Ucp::Category.new(value: type, taxonomy: "merchant")]
|
|
121
|
+
end
|
|
122
|
+
end
|
|
123
|
+
end
|
|
124
|
+
end
|
|
125
|
+
end
|
|
126
|
+
end
|
|
127
|
+
end
|
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
require "net/http"
|
|
2
|
+
require "uri"
|
|
3
|
+
require "json"
|
|
4
|
+
require "portage/ucp/support/connection"
|
|
5
|
+
|
|
6
|
+
require_relative "../../../user_agent"
|
|
7
|
+
|
|
8
|
+
module Portage
|
|
9
|
+
module Cli
|
|
10
|
+
module Index
|
|
11
|
+
module Sources
|
|
12
|
+
class StorefrontProducts
|
|
13
|
+
# The HTTP half of a crawl: GETs `products.json?limit=&page=`
|
|
14
|
+
# through Support::Connection with the shared UserAgent, pausing
|
|
15
|
+
# between pages, waiting out one 429's Retry-After and stopping on
|
|
16
|
+
# a second, and turning anything that isn't a products array into
|
|
17
|
+
# a reason.
|
|
18
|
+
class Pages
|
|
19
|
+
PAUSE = 1
|
|
20
|
+
MAX_RETRY_AFTER = 60
|
|
21
|
+
DEFAULT_RETRY_AFTER = 10
|
|
22
|
+
OPEN_TIMEOUT = 5
|
|
23
|
+
READ_TIMEOUT = 20
|
|
24
|
+
|
|
25
|
+
def self.get(url, accept: "application/json")
|
|
26
|
+
uri = URI.parse(url)
|
|
27
|
+
headers = UserAgent.headers.merge("Accept" => accept)
|
|
28
|
+
Portage::Ucp::Support::Connection.start(uri, route: :store, open_timeout: OPEN_TIMEOUT,
|
|
29
|
+
read_timeout: READ_TIMEOUT) do |http|
|
|
30
|
+
http.get(uri.request_uri, headers)
|
|
31
|
+
end
|
|
32
|
+
end
|
|
33
|
+
|
|
34
|
+
# @param allowed [#call] path -> whether robots.txt allows it.
|
|
35
|
+
def initialize(endpoint, per_page:, max_pages:, sleeper:, allowed: ->(_path) { true })
|
|
36
|
+
@endpoint = endpoint
|
|
37
|
+
@path = URI.parse(endpoint).path
|
|
38
|
+
@allowed = allowed
|
|
39
|
+
@per_page = per_page
|
|
40
|
+
@max_pages = max_pages
|
|
41
|
+
@sleeper = sleeper
|
|
42
|
+
@rate_limited = 0
|
|
43
|
+
end
|
|
44
|
+
|
|
45
|
+
# Yields each non-empty page's raw products.
|
|
46
|
+
# @return [Array(String, String, Integer)] status ("ok",
|
|
47
|
+
# "partial", "skipped"), reason (nil when ok) and pages read.
|
|
48
|
+
def each(&)
|
|
49
|
+
(1..@max_pages).each do |page|
|
|
50
|
+
@sleeper.call(PAUSE) if page > 1
|
|
51
|
+
products = fetch(page)
|
|
52
|
+
done = outcome(products, page, &)
|
|
53
|
+
return done if done
|
|
54
|
+
end
|
|
55
|
+
["partial", "page_cap", @max_pages]
|
|
56
|
+
end
|
|
57
|
+
|
|
58
|
+
private
|
|
59
|
+
|
|
60
|
+
# nil means "full page, keep going".
|
|
61
|
+
def outcome(products, page)
|
|
62
|
+
return failed(products, page - 1) unless products.is_a?(Array)
|
|
63
|
+
return ["skipped", "empty", 0] if products.empty? && page == 1
|
|
64
|
+
|
|
65
|
+
yield products unless products.empty?
|
|
66
|
+
["ok", nil, page] if products.length < @per_page
|
|
67
|
+
end
|
|
68
|
+
|
|
69
|
+
# Past page 1 a failure keeps what earlier pages found.
|
|
70
|
+
def failed(reason, pages) = [pages.zero? ? "skipped" : "partial", reason, pages]
|
|
71
|
+
|
|
72
|
+
# @return [Array<Hash>, String] a page's products, or why there
|
|
73
|
+
# are none.
|
|
74
|
+
def fetch(page)
|
|
75
|
+
return "robots" unless @allowed.call("#{@path}#{query(page)}")
|
|
76
|
+
|
|
77
|
+
response = self.class.get("#{@endpoint}#{query(page)}")
|
|
78
|
+
return products_in(response) unless response.code == "429"
|
|
79
|
+
|
|
80
|
+
@rate_limited += 1
|
|
81
|
+
return "rate_limited" if @rate_limited > 1
|
|
82
|
+
|
|
83
|
+
@sleeper.call(retry_after(response["Retry-After"]))
|
|
84
|
+
fetch(page)
|
|
85
|
+
rescue StandardError => e
|
|
86
|
+
"error: #{e.class}"
|
|
87
|
+
end
|
|
88
|
+
|
|
89
|
+
# Page 1 is the bare URL (Shopify's default page), so a robots
|
|
90
|
+
# rule aimed at duplicate `?page=1` URLs doesn't read as a ban
|
|
91
|
+
# on the whole catalogue.
|
|
92
|
+
def query(page) = page == 1 ? "?limit=#{@per_page}" : "?limit=#{@per_page}&page=#{page}"
|
|
93
|
+
|
|
94
|
+
def products_in(response)
|
|
95
|
+
return "not_found" if response.code == "404"
|
|
96
|
+
return "redirect" if response.is_a?(Net::HTTPRedirection)
|
|
97
|
+
return "http_#{response.code}" unless response.is_a?(Net::HTTPSuccess)
|
|
98
|
+
|
|
99
|
+
body = JSON.parse(response.body.to_s.dup.force_encoding(Encoding::UTF_8))
|
|
100
|
+
body.is_a?(Hash) && body["products"].is_a?(Array) ? body["products"] : "not_json"
|
|
101
|
+
rescue JSON::ParserError
|
|
102
|
+
"not_json"
|
|
103
|
+
end
|
|
104
|
+
|
|
105
|
+
def retry_after(value)
|
|
106
|
+
seconds = Integer(value.to_s, exception: false) || DEFAULT_RETRY_AFTER
|
|
107
|
+
seconds.clamp(0, MAX_RETRY_AFTER)
|
|
108
|
+
end
|
|
109
|
+
end
|
|
110
|
+
end
|
|
111
|
+
end
|
|
112
|
+
end
|
|
113
|
+
end
|
|
114
|
+
end
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
module Portage
|
|
2
|
+
module Cli
|
|
3
|
+
module Index
|
|
4
|
+
module Sources
|
|
5
|
+
class StorefrontProducts
|
|
6
|
+
# Just enough robots.txt (RFC 9309) to answer "may this agent GET
|
|
7
|
+
# this path?": the group naming this agent if there is one, else
|
|
8
|
+
# the `*` group; `*` and `$` in rules; the longest matching rule
|
|
9
|
+
# wins, and Allow wins a tie. No rules, or no file, means allowed.
|
|
10
|
+
class Robots
|
|
11
|
+
def initialize(body)
|
|
12
|
+
@groups = parse(body.to_s)
|
|
13
|
+
end
|
|
14
|
+
|
|
15
|
+
# @param path [String] request path, query included.
|
|
16
|
+
# @param agent [String] the User-Agent this process sends.
|
|
17
|
+
def allowed?(path, agent:)
|
|
18
|
+
rules = rules_for(agent.to_s.downcase)
|
|
19
|
+
matches = rules.select { |_allow, pattern| match?(pattern, path) }
|
|
20
|
+
return true if matches.empty?
|
|
21
|
+
|
|
22
|
+
longest = matches.map { |_allow, pattern| pattern.length }.max
|
|
23
|
+
matches.select { |_allow, pattern| pattern.length == longest }.any?(&:first)
|
|
24
|
+
end
|
|
25
|
+
|
|
26
|
+
private
|
|
27
|
+
|
|
28
|
+
def rules_for(agent)
|
|
29
|
+
named = @groups.select { |agents, *| agents.any? { |a| a != "*" && agent.include?(a) } }
|
|
30
|
+
chosen = named.empty? ? @groups.select { |agents, *| agents.include?("*") } : named
|
|
31
|
+
chosen.flat_map { |_agents, rules, _closed| rules }
|
|
32
|
+
end
|
|
33
|
+
|
|
34
|
+
# @return [Array<Array>] [agents, [[allow?, pattern], ...], closed]
|
|
35
|
+
# per group.
|
|
36
|
+
def parse(body)
|
|
37
|
+
groups = []
|
|
38
|
+
body.each_line do |line|
|
|
39
|
+
field, value = line.sub(/#.*/, "").split(":", 2).map { |part| part.to_s.strip }
|
|
40
|
+
next if value.nil?
|
|
41
|
+
|
|
42
|
+
add_line(groups, field.downcase, value)
|
|
43
|
+
end
|
|
44
|
+
groups
|
|
45
|
+
end
|
|
46
|
+
|
|
47
|
+
# A rule line (even an empty `Disallow:`) ends a group's
|
|
48
|
+
# User-agent list, so the next User-agent starts a new group.
|
|
49
|
+
def add_line(groups, field, value)
|
|
50
|
+
if field == "user-agent"
|
|
51
|
+
groups << [[], [], false] if groups.empty? || groups.last[2]
|
|
52
|
+
groups.last[0] << value.downcase
|
|
53
|
+
elsif %w[allow disallow].include?(field) && groups.any?
|
|
54
|
+
groups.last[2] = true
|
|
55
|
+
groups.last[1] << [field == "allow", value] unless value.empty?
|
|
56
|
+
end
|
|
57
|
+
end
|
|
58
|
+
|
|
59
|
+
def match?(pattern, path)
|
|
60
|
+
anchored = pattern.end_with?("$")
|
|
61
|
+
body = anchored ? pattern.chomp("$") : pattern
|
|
62
|
+
regex = body.split("*", -1).map { |part| Regexp.escape(part) }.join(".*")
|
|
63
|
+
Regexp.new("\\A#{regex}#{'\\z' if anchored}").match?(path)
|
|
64
|
+
end
|
|
65
|
+
end
|
|
66
|
+
end
|
|
67
|
+
end
|
|
68
|
+
end
|
|
69
|
+
end
|
|
70
|
+
end
|
|
@@ -0,0 +1,147 @@
|
|
|
1
|
+
require "net/http"
|
|
2
|
+
require "uri"
|
|
3
|
+
|
|
4
|
+
require_relative "../../user_agent"
|
|
5
|
+
require_relative "../../classifier"
|
|
6
|
+
require_relative "../../handoff_only"
|
|
7
|
+
require_relative "../store"
|
|
8
|
+
require_relative "storefront_products/mapper"
|
|
9
|
+
require_relative "storefront_products/robots"
|
|
10
|
+
require_relative "storefront_products/pages"
|
|
11
|
+
|
|
12
|
+
module Portage
|
|
13
|
+
module Cli
|
|
14
|
+
module Index
|
|
15
|
+
module Sources
|
|
16
|
+
# A store's own public catalogue, read from Shopify's `/products.json`
|
|
17
|
+
# (docs/plans/local-catalogue.md Phase 2), for origins already in
|
|
18
|
+
# the index. Off by default: it runs only from `portage index build
|
|
19
|
+
# --sources storefront_products` or `portage index add URL --crawl`,
|
|
20
|
+
# never from `find` or `buy`.
|
|
21
|
+
#
|
|
22
|
+
# Guardrails: a hand-off-only host (HandoffOnly, checked live) and a
|
|
23
|
+
# robots.txt Disallow get no products.json request at all; pages are
|
|
24
|
+
# MAX_PAGES x 250 at most, PAUSE seconds apart; a 429 waits out its
|
|
25
|
+
# Retry-After once and a second 429 stops that store; a 404, a
|
|
26
|
+
# redirect, an empty page 1, or anything that isn't products JSON
|
|
27
|
+
# (a bot wall) skips the store. Every outcome is noted on the store
|
|
28
|
+
# row as `crawl` ({status, reason, pages, products, at}).
|
|
29
|
+
#
|
|
30
|
+
# Yields Mapper sightings (no price, no availability) plus one store
|
|
31
|
+
# sighting per origin carrying `store_fields:` (the crawl note, and
|
|
32
|
+
# platform "shopify" when products.json answered); Index::Builder
|
|
33
|
+
# writes both.
|
|
34
|
+
class StorefrontProducts
|
|
35
|
+
MAX_PAGES = 20
|
|
36
|
+
MAX_STORES = 25
|
|
37
|
+
PER_PAGE = 250
|
|
38
|
+
|
|
39
|
+
# The one platform seam: WooCommerce's Store API can join here.
|
|
40
|
+
# nil platform means "not known yet", worth a try.
|
|
41
|
+
def self.endpoint_for(origin, platform)
|
|
42
|
+
"#{origin}/products.json" if platform.nil? || platform == "shopify"
|
|
43
|
+
end
|
|
44
|
+
|
|
45
|
+
# The guardrails, plus test seams for sleeping and the clock.
|
|
46
|
+
def initialize(stores: Store.new, handoff_only: HandoffOnly.new, max_pages: MAX_PAGES,
|
|
47
|
+
max_stores: MAX_STORES, per_page: PER_PAGE, sleeper: Kernel.method(:sleep), now: Time.now)
|
|
48
|
+
@stores = stores
|
|
49
|
+
@handoff_only = handoff_only
|
|
50
|
+
@max_pages = max_pages
|
|
51
|
+
@max_stores = max_stores
|
|
52
|
+
@per_page = per_page
|
|
53
|
+
@sleeper = sleeper
|
|
54
|
+
@now = now
|
|
55
|
+
end
|
|
56
|
+
|
|
57
|
+
def name = "storefront_products"
|
|
58
|
+
|
|
59
|
+
def description
|
|
60
|
+
"Each indexed store's own /products.json (Shopify), up to #{MAX_PAGES} pages a store and " \
|
|
61
|
+
"#{MAX_STORES} stores a run, least recently crawled first. Off by default — opt in with " \
|
|
62
|
+
"--sources storefront_products or `index add URL --crawl`."
|
|
63
|
+
end
|
|
64
|
+
|
|
65
|
+
def source_path = nil
|
|
66
|
+
|
|
67
|
+
# `**` accepts (and ignores) the shared Source#candidates(queries:)
|
|
68
|
+
# interface.
|
|
69
|
+
def candidates(**)
|
|
70
|
+
eligible = @stores.all.select do |entry|
|
|
71
|
+
self.class.endpoint_for(entry["origin"], entry["platform"]) && !handoff_only?(entry["origin"])
|
|
72
|
+
end
|
|
73
|
+
eligible.sort_by { |entry| entry.dig("crawl", "at").to_i }.first(@max_stores)
|
|
74
|
+
.flat_map { |entry| crawl(entry["origin"], platform: entry["platform"]) }
|
|
75
|
+
end
|
|
76
|
+
|
|
77
|
+
# @return [Array<Hash>] product sightings, then the store sighting.
|
|
78
|
+
def crawl(origin, platform: nil)
|
|
79
|
+
endpoint = self.class.endpoint_for(origin, platform)
|
|
80
|
+
robots = robots_for(origin, endpoint)
|
|
81
|
+
return [store_sighting(origin, note("skipped", robots, 0, 0))] if robots.is_a?(String)
|
|
82
|
+
|
|
83
|
+
sightings = []
|
|
84
|
+
classify = memoized_classifier
|
|
85
|
+
status, reason, pages = pages_for(endpoint, robots).each do |products|
|
|
86
|
+
sightings.concat(products.map { |raw| Mapper.sighting(raw, origin: origin, classify: classify) })
|
|
87
|
+
end
|
|
88
|
+
sightings + [store_sighting(origin, note(status, reason, pages, sightings.length))]
|
|
89
|
+
rescue StandardError => e
|
|
90
|
+
[store_sighting(origin, note("skipped", "error: #{e.class}", 0, 0))]
|
|
91
|
+
end
|
|
92
|
+
|
|
93
|
+
private
|
|
94
|
+
|
|
95
|
+
def pages_for(endpoint, robots)
|
|
96
|
+
Pages.new(endpoint, per_page: @per_page, max_pages: @max_pages, sleeper: @sleeper,
|
|
97
|
+
allowed: ->(path) { robots.allowed?(path, agent: UserAgent.value) })
|
|
98
|
+
end
|
|
99
|
+
|
|
100
|
+
# The store's robots.txt rules, or why the crawl stops before
|
|
101
|
+
# any request (no endpoint, hand-off only) or before any
|
|
102
|
+
# products.json request (robots.txt unreachable). Pages checks
|
|
103
|
+
# the rules against every page URL it would request.
|
|
104
|
+
# RFC 9309: a 4xx robots.txt means no rules; a 5xx or no answer
|
|
105
|
+
# at all means stay out.
|
|
106
|
+
# @return [Robots, String]
|
|
107
|
+
def robots_for(origin, endpoint)
|
|
108
|
+
return "unsupported_platform" unless endpoint
|
|
109
|
+
return "handoff_only" if handoff_only?(origin)
|
|
110
|
+
|
|
111
|
+
response = Pages.get("#{origin}/robots.txt", accept: "text/plain")
|
|
112
|
+
return Robots.new(nil) if response.code.start_with?("4")
|
|
113
|
+
return "redirect" if response.is_a?(Net::HTTPRedirection)
|
|
114
|
+
return "robots_unreachable" unless response.is_a?(Net::HTTPSuccess)
|
|
115
|
+
|
|
116
|
+
Robots.new(response.body.to_s.dup.force_encoding(Encoding::UTF_8))
|
|
117
|
+
rescue StandardError
|
|
118
|
+
"robots_unreachable"
|
|
119
|
+
end
|
|
120
|
+
|
|
121
|
+
# Product types and tags repeat across a catalogue, and Classifier
|
|
122
|
+
# re-reads its YAML on every call.
|
|
123
|
+
def memoized_classifier
|
|
124
|
+
memo = {}
|
|
125
|
+
->(text) { memo[text] ||= Classifier.categories_for(text) }
|
|
126
|
+
end
|
|
127
|
+
|
|
128
|
+
def handoff_only?(origin)
|
|
129
|
+
@handoff_only.host?(URI.parse(origin.to_s).host)
|
|
130
|
+
rescue URI::InvalidURIError
|
|
131
|
+
true
|
|
132
|
+
end
|
|
133
|
+
|
|
134
|
+
def note(status, reason, pages, products)
|
|
135
|
+
{ "status" => status, "reason" => reason, "pages" => pages, "products" => products, "at" => @now.to_i }
|
|
136
|
+
end
|
|
137
|
+
|
|
138
|
+
def store_sighting(origin, crawl)
|
|
139
|
+
fields = { crawl: crawl }
|
|
140
|
+
fields[:platform] = "shopify" unless crawl["status"] == "skipped"
|
|
141
|
+
{ origin: origin, url: nil, title: nil, brand: nil, gtin: nil, store_fields: fields }
|
|
142
|
+
end
|
|
143
|
+
end
|
|
144
|
+
end
|
|
145
|
+
end
|
|
146
|
+
end
|
|
147
|
+
end
|
|
@@ -3,6 +3,7 @@ require_relative "sources/stores_file"
|
|
|
3
3
|
require_relative "sources/browser"
|
|
4
4
|
require_relative "sources/wikidata"
|
|
5
5
|
require_relative "sources/webmcp_sweep"
|
|
6
|
+
require_relative "sources/storefront_products"
|
|
6
7
|
|
|
7
8
|
module Portage
|
|
8
9
|
module Cli
|
|
@@ -16,7 +17,8 @@ module Portage
|
|
|
16
17
|
"stores_file" => -> { StoresFile.new },
|
|
17
18
|
"browser" => -> { Browser.new },
|
|
18
19
|
"wikidata" => -> { Wikidata.new },
|
|
19
|
-
"webmcp_sweep" => -> { WebmcpSweep.new }
|
|
20
|
+
"webmcp_sweep" => -> { WebmcpSweep.new },
|
|
21
|
+
"storefront_products" => -> { StorefrontProducts.new }
|
|
20
22
|
}.freeze
|
|
21
23
|
|
|
22
24
|
# Runs with no extra opt-in: no bridge required, and no low-yield
|
|
@@ -24,7 +26,8 @@ module Portage
|
|
|
24
26
|
# show up in `portage index sources`, just not run unless named in
|
|
25
27
|
# `--sources`. `browser` is listed but never a default — it yields
|
|
26
28
|
# nothing; `portage browser import` writes those entries itself
|
|
27
|
-
# (see Sources::Browser).
|
|
29
|
+
# (see Sources::Browser). `storefront_products` sends up to 21
|
|
30
|
+
# requests a store, so it runs only when named.
|
|
28
31
|
DEFAULT_NAMES = %w[shopify_catalog stores_file].freeze
|
|
29
32
|
|
|
30
33
|
def self.all = ALL.values.map(&:call)
|