portage-cli 0.10.0 → 0.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +129 -0
- data/README.md +107 -21
- data/known-stores/categories.yml +6841 -18
- data/known-stores/category-stoplist.yml +40 -0
- data/known-stores/category-synonyms.yml +15 -0
- data/lib/portage/cli/browser_import/categorize.rb +5 -2
- data/lib/portage/cli/buy.rb +217 -39
- data/lib/portage/cli/check.rb +11 -1
- data/lib/portage/cli/classifier/ranking.rb +135 -0
- data/lib/portage/cli/classifier/table.rb +63 -0
- data/lib/portage/cli/classifier.rb +23 -44
- data/lib/portage/cli/confidence_check.rb +27 -7
- data/lib/portage/cli/confidence_state.rb +109 -0
- data/lib/portage/cli/doctor.rb +16 -6
- data/lib/portage/cli/find.rb +69 -4
- data/lib/portage/cli/index/builder.rb +49 -12
- data/lib/portage/cli/index/database.rb +150 -0
- data/lib/portage/cli/index/entry_product.rb +37 -0
- data/lib/portage/cli/index/legacy_import.rb +54 -0
- data/lib/portage/cli/index/product_store.rb +73 -35
- data/lib/portage/cli/index/schema.rb +70 -0
- data/lib/portage/cli/index/search.rb +73 -0
- data/lib/portage/cli/index/sources/storefront_products/mapper.rb +127 -0
- data/lib/portage/cli/index/sources/storefront_products/pages.rb +114 -0
- data/lib/portage/cli/index/sources/storefront_products/robots.rb +70 -0
- data/lib/portage/cli/index/sources/storefront_products.rb +147 -0
- data/lib/portage/cli/index/sources.rb +5 -2
- data/lib/portage/cli/index/store.rb +23 -47
- data/lib/portage/cli/index.rb +1 -0
- data/lib/portage/cli/offer_sources.rb +17 -2
- data/lib/portage/cli/version.rb +1 -1
- data/lib/portage/cli.rb +111 -12
- metadata +30 -2
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
require "net/http"
|
|
2
|
+
require "uri"
|
|
3
|
+
require "json"
|
|
4
|
+
require "portage/ucp/support/connection"
|
|
5
|
+
|
|
6
|
+
require_relative "../../../user_agent"
|
|
7
|
+
|
|
8
|
+
module Portage
|
|
9
|
+
module Cli
|
|
10
|
+
module Index
|
|
11
|
+
module Sources
|
|
12
|
+
class StorefrontProducts
|
|
13
|
+
# The HTTP half of a crawl: GETs `products.json?limit=&page=`
|
|
14
|
+
# through Support::Connection with the shared UserAgent, pausing
|
|
15
|
+
# between pages, waiting out one 429's Retry-After and stopping on
|
|
16
|
+
# a second, and turning anything that isn't a products array into
|
|
17
|
+
# a reason.
|
|
18
|
+
class Pages
|
|
19
|
+
PAUSE = 1
|
|
20
|
+
MAX_RETRY_AFTER = 60
|
|
21
|
+
DEFAULT_RETRY_AFTER = 10
|
|
22
|
+
OPEN_TIMEOUT = 5
|
|
23
|
+
READ_TIMEOUT = 20
|
|
24
|
+
|
|
25
|
+
def self.get(url, accept: "application/json")
|
|
26
|
+
uri = URI.parse(url)
|
|
27
|
+
headers = UserAgent.headers.merge("Accept" => accept)
|
|
28
|
+
Portage::Ucp::Support::Connection.start(uri, route: :store, open_timeout: OPEN_TIMEOUT,
|
|
29
|
+
read_timeout: READ_TIMEOUT) do |http|
|
|
30
|
+
http.get(uri.request_uri, headers)
|
|
31
|
+
end
|
|
32
|
+
end
|
|
33
|
+
|
|
34
|
+
# @param allowed [#call] path -> whether robots.txt allows it.
|
|
35
|
+
def initialize(endpoint, per_page:, max_pages:, sleeper:, allowed: ->(_path) { true })
|
|
36
|
+
@endpoint = endpoint
|
|
37
|
+
@path = URI.parse(endpoint).path
|
|
38
|
+
@allowed = allowed
|
|
39
|
+
@per_page = per_page
|
|
40
|
+
@max_pages = max_pages
|
|
41
|
+
@sleeper = sleeper
|
|
42
|
+
@rate_limited = 0
|
|
43
|
+
end
|
|
44
|
+
|
|
45
|
+
# Yields each non-empty page's raw products.
|
|
46
|
+
# @return [Array(String, String, Integer)] status ("ok",
|
|
47
|
+
# "partial", "skipped"), reason (nil when ok) and pages read.
|
|
48
|
+
def each(&)
|
|
49
|
+
(1..@max_pages).each do |page|
|
|
50
|
+
@sleeper.call(PAUSE) if page > 1
|
|
51
|
+
products = fetch(page)
|
|
52
|
+
done = outcome(products, page, &)
|
|
53
|
+
return done if done
|
|
54
|
+
end
|
|
55
|
+
["partial", "page_cap", @max_pages]
|
|
56
|
+
end
|
|
57
|
+
|
|
58
|
+
private
|
|
59
|
+
|
|
60
|
+
# nil means "full page, keep going".
|
|
61
|
+
def outcome(products, page)
|
|
62
|
+
return failed(products, page - 1) unless products.is_a?(Array)
|
|
63
|
+
return ["skipped", "empty", 0] if products.empty? && page == 1
|
|
64
|
+
|
|
65
|
+
yield products unless products.empty?
|
|
66
|
+
["ok", nil, page] if products.length < @per_page
|
|
67
|
+
end
|
|
68
|
+
|
|
69
|
+
# Past page 1 a failure keeps what earlier pages found.
|
|
70
|
+
def failed(reason, pages) = [pages.zero? ? "skipped" : "partial", reason, pages]
|
|
71
|
+
|
|
72
|
+
# @return [Array<Hash>, String] a page's products, or why there
|
|
73
|
+
# are none.
|
|
74
|
+
def fetch(page)
|
|
75
|
+
return "robots" unless @allowed.call("#{@path}#{query(page)}")
|
|
76
|
+
|
|
77
|
+
response = self.class.get("#{@endpoint}#{query(page)}")
|
|
78
|
+
return products_in(response) unless response.code == "429"
|
|
79
|
+
|
|
80
|
+
@rate_limited += 1
|
|
81
|
+
return "rate_limited" if @rate_limited > 1
|
|
82
|
+
|
|
83
|
+
@sleeper.call(retry_after(response["Retry-After"]))
|
|
84
|
+
fetch(page)
|
|
85
|
+
rescue StandardError => e
|
|
86
|
+
"error: #{e.class}"
|
|
87
|
+
end
|
|
88
|
+
|
|
89
|
+
# Page 1 is the bare URL (Shopify's default page), so a robots
|
|
90
|
+
# rule aimed at duplicate `?page=1` URLs doesn't read as a ban
|
|
91
|
+
# on the whole catalogue.
|
|
92
|
+
def query(page) = page == 1 ? "?limit=#{@per_page}" : "?limit=#{@per_page}&page=#{page}"
|
|
93
|
+
|
|
94
|
+
def products_in(response)
|
|
95
|
+
return "not_found" if response.code == "404"
|
|
96
|
+
return "redirect" if response.is_a?(Net::HTTPRedirection)
|
|
97
|
+
return "http_#{response.code}" unless response.is_a?(Net::HTTPSuccess)
|
|
98
|
+
|
|
99
|
+
body = JSON.parse(response.body.to_s.dup.force_encoding(Encoding::UTF_8))
|
|
100
|
+
body.is_a?(Hash) && body["products"].is_a?(Array) ? body["products"] : "not_json"
|
|
101
|
+
rescue JSON::ParserError
|
|
102
|
+
"not_json"
|
|
103
|
+
end
|
|
104
|
+
|
|
105
|
+
def retry_after(value)
|
|
106
|
+
seconds = Integer(value.to_s, exception: false) || DEFAULT_RETRY_AFTER
|
|
107
|
+
seconds.clamp(0, MAX_RETRY_AFTER)
|
|
108
|
+
end
|
|
109
|
+
end
|
|
110
|
+
end
|
|
111
|
+
end
|
|
112
|
+
end
|
|
113
|
+
end
|
|
114
|
+
end
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
module Portage
|
|
2
|
+
module Cli
|
|
3
|
+
module Index
|
|
4
|
+
module Sources
|
|
5
|
+
class StorefrontProducts
|
|
6
|
+
# Just enough robots.txt (RFC 9309) to answer "may this agent GET
|
|
7
|
+
# this path?": the group naming this agent if there is one, else
|
|
8
|
+
# the `*` group; `*` and `$` in rules; the longest matching rule
|
|
9
|
+
# wins, and Allow wins a tie. No rules, or no file, means allowed.
|
|
10
|
+
class Robots
|
|
11
|
+
def initialize(body)
|
|
12
|
+
@groups = parse(body.to_s)
|
|
13
|
+
end
|
|
14
|
+
|
|
15
|
+
# @param path [String] request path, query included.
|
|
16
|
+
# @param agent [String] the User-Agent this process sends.
|
|
17
|
+
def allowed?(path, agent:)
|
|
18
|
+
rules = rules_for(agent.to_s.downcase)
|
|
19
|
+
matches = rules.select { |_allow, pattern| match?(pattern, path) }
|
|
20
|
+
return true if matches.empty?
|
|
21
|
+
|
|
22
|
+
longest = matches.map { |_allow, pattern| pattern.length }.max
|
|
23
|
+
matches.select { |_allow, pattern| pattern.length == longest }.any?(&:first)
|
|
24
|
+
end
|
|
25
|
+
|
|
26
|
+
private
|
|
27
|
+
|
|
28
|
+
def rules_for(agent)
|
|
29
|
+
named = @groups.select { |agents, *| agents.any? { |a| a != "*" && agent.include?(a) } }
|
|
30
|
+
chosen = named.empty? ? @groups.select { |agents, *| agents.include?("*") } : named
|
|
31
|
+
chosen.flat_map { |_agents, rules, _closed| rules }
|
|
32
|
+
end
|
|
33
|
+
|
|
34
|
+
# @return [Array<Array>] [agents, [[allow?, pattern], ...], closed]
|
|
35
|
+
# per group.
|
|
36
|
+
def parse(body)
|
|
37
|
+
groups = []
|
|
38
|
+
body.each_line do |line|
|
|
39
|
+
field, value = line.sub(/#.*/, "").split(":", 2).map { |part| part.to_s.strip }
|
|
40
|
+
next if value.nil?
|
|
41
|
+
|
|
42
|
+
add_line(groups, field.downcase, value)
|
|
43
|
+
end
|
|
44
|
+
groups
|
|
45
|
+
end
|
|
46
|
+
|
|
47
|
+
# A rule line (even an empty `Disallow:`) ends a group's
|
|
48
|
+
# User-agent list, so the next User-agent starts a new group.
|
|
49
|
+
def add_line(groups, field, value)
|
|
50
|
+
if field == "user-agent"
|
|
51
|
+
groups << [[], [], false] if groups.empty? || groups.last[2]
|
|
52
|
+
groups.last[0] << value.downcase
|
|
53
|
+
elsif %w[allow disallow].include?(field) && groups.any?
|
|
54
|
+
groups.last[2] = true
|
|
55
|
+
groups.last[1] << [field == "allow", value] unless value.empty?
|
|
56
|
+
end
|
|
57
|
+
end
|
|
58
|
+
|
|
59
|
+
def match?(pattern, path)
|
|
60
|
+
anchored = pattern.end_with?("$")
|
|
61
|
+
body = anchored ? pattern.chomp("$") : pattern
|
|
62
|
+
regex = body.split("*", -1).map { |part| Regexp.escape(part) }.join(".*")
|
|
63
|
+
Regexp.new("\\A#{regex}#{'\\z' if anchored}").match?(path)
|
|
64
|
+
end
|
|
65
|
+
end
|
|
66
|
+
end
|
|
67
|
+
end
|
|
68
|
+
end
|
|
69
|
+
end
|
|
70
|
+
end
|
|
@@ -0,0 +1,147 @@
|
|
|
1
|
+
require "net/http"
|
|
2
|
+
require "uri"
|
|
3
|
+
|
|
4
|
+
require_relative "../../user_agent"
|
|
5
|
+
require_relative "../../classifier"
|
|
6
|
+
require_relative "../../handoff_only"
|
|
7
|
+
require_relative "../store"
|
|
8
|
+
require_relative "storefront_products/mapper"
|
|
9
|
+
require_relative "storefront_products/robots"
|
|
10
|
+
require_relative "storefront_products/pages"
|
|
11
|
+
|
|
12
|
+
module Portage
|
|
13
|
+
module Cli
|
|
14
|
+
module Index
|
|
15
|
+
module Sources
|
|
16
|
+
# A store's own public catalogue, read from Shopify's `/products.json`
|
|
17
|
+
# (docs/plans/local-catalogue.md Phase 2), for origins already in
|
|
18
|
+
# the index. Off by default: it runs only from `portage index build
|
|
19
|
+
# --sources storefront_products` or `portage index add URL --crawl`,
|
|
20
|
+
# never from `find` or `buy`.
|
|
21
|
+
#
|
|
22
|
+
# Guardrails: a hand-off-only host (HandoffOnly, checked live) and a
|
|
23
|
+
# robots.txt Disallow get no products.json request at all; pages are
|
|
24
|
+
# MAX_PAGES x 250 at most, PAUSE seconds apart; a 429 waits out its
|
|
25
|
+
# Retry-After once and a second 429 stops that store; a 404, a
|
|
26
|
+
# redirect, an empty page 1, or anything that isn't products JSON
|
|
27
|
+
# (a bot wall) skips the store. Every outcome is noted on the store
|
|
28
|
+
# row as `crawl` ({status, reason, pages, products, at}).
|
|
29
|
+
#
|
|
30
|
+
# Yields Mapper sightings (no price, no availability) plus one store
|
|
31
|
+
# sighting per origin carrying `store_fields:` (the crawl note, and
|
|
32
|
+
# platform "shopify" when products.json answered); Index::Builder
|
|
33
|
+
# writes both.
|
|
34
|
+
class StorefrontProducts
|
|
35
|
+
MAX_PAGES = 20
|
|
36
|
+
MAX_STORES = 25
|
|
37
|
+
PER_PAGE = 250
|
|
38
|
+
|
|
39
|
+
# The one platform seam: WooCommerce's Store API can join here.
|
|
40
|
+
# nil platform means "not known yet", worth a try.
|
|
41
|
+
def self.endpoint_for(origin, platform)
|
|
42
|
+
"#{origin}/products.json" if platform.nil? || platform == "shopify"
|
|
43
|
+
end
|
|
44
|
+
|
|
45
|
+
# The guardrails, plus test seams for sleeping and the clock.
|
|
46
|
+
def initialize(stores: Store.new, handoff_only: HandoffOnly.new, max_pages: MAX_PAGES,
|
|
47
|
+
max_stores: MAX_STORES, per_page: PER_PAGE, sleeper: Kernel.method(:sleep), now: Time.now)
|
|
48
|
+
@stores = stores
|
|
49
|
+
@handoff_only = handoff_only
|
|
50
|
+
@max_pages = max_pages
|
|
51
|
+
@max_stores = max_stores
|
|
52
|
+
@per_page = per_page
|
|
53
|
+
@sleeper = sleeper
|
|
54
|
+
@now = now
|
|
55
|
+
end
|
|
56
|
+
|
|
57
|
+
def name = "storefront_products"
|
|
58
|
+
|
|
59
|
+
def description
|
|
60
|
+
"Each indexed store's own /products.json (Shopify), up to #{MAX_PAGES} pages a store and " \
|
|
61
|
+
"#{MAX_STORES} stores a run, least recently crawled first. Off by default — opt in with " \
|
|
62
|
+
"--sources storefront_products or `index add URL --crawl`."
|
|
63
|
+
end
|
|
64
|
+
|
|
65
|
+
def source_path = nil
|
|
66
|
+
|
|
67
|
+
# `**` accepts (and ignores) the shared Source#candidates(queries:)
|
|
68
|
+
# interface.
|
|
69
|
+
def candidates(**)
|
|
70
|
+
eligible = @stores.all.select do |entry|
|
|
71
|
+
self.class.endpoint_for(entry["origin"], entry["platform"]) && !handoff_only?(entry["origin"])
|
|
72
|
+
end
|
|
73
|
+
eligible.sort_by { |entry| entry.dig("crawl", "at").to_i }.first(@max_stores)
|
|
74
|
+
.flat_map { |entry| crawl(entry["origin"], platform: entry["platform"]) }
|
|
75
|
+
end
|
|
76
|
+
|
|
77
|
+
# @return [Array<Hash>] product sightings, then the store sighting.
|
|
78
|
+
def crawl(origin, platform: nil)
|
|
79
|
+
endpoint = self.class.endpoint_for(origin, platform)
|
|
80
|
+
robots = robots_for(origin, endpoint)
|
|
81
|
+
return [store_sighting(origin, note("skipped", robots, 0, 0))] if robots.is_a?(String)
|
|
82
|
+
|
|
83
|
+
sightings = []
|
|
84
|
+
classify = memoized_classifier
|
|
85
|
+
status, reason, pages = pages_for(endpoint, robots).each do |products|
|
|
86
|
+
sightings.concat(products.map { |raw| Mapper.sighting(raw, origin: origin, classify: classify) })
|
|
87
|
+
end
|
|
88
|
+
sightings + [store_sighting(origin, note(status, reason, pages, sightings.length))]
|
|
89
|
+
rescue StandardError => e
|
|
90
|
+
[store_sighting(origin, note("skipped", "error: #{e.class}", 0, 0))]
|
|
91
|
+
end
|
|
92
|
+
|
|
93
|
+
private
|
|
94
|
+
|
|
95
|
+
def pages_for(endpoint, robots)
|
|
96
|
+
Pages.new(endpoint, per_page: @per_page, max_pages: @max_pages, sleeper: @sleeper,
|
|
97
|
+
allowed: ->(path) { robots.allowed?(path, agent: UserAgent.value) })
|
|
98
|
+
end
|
|
99
|
+
|
|
100
|
+
# The store's robots.txt rules, or why the crawl stops before
|
|
101
|
+
# any request (no endpoint, hand-off only) or before any
|
|
102
|
+
# products.json request (robots.txt unreachable). Pages checks
|
|
103
|
+
# the rules against every page URL it would request.
|
|
104
|
+
# RFC 9309: a 4xx robots.txt means no rules; a 5xx or no answer
|
|
105
|
+
# at all means stay out.
|
|
106
|
+
# @return [Robots, String]
|
|
107
|
+
def robots_for(origin, endpoint)
|
|
108
|
+
return "unsupported_platform" unless endpoint
|
|
109
|
+
return "handoff_only" if handoff_only?(origin)
|
|
110
|
+
|
|
111
|
+
response = Pages.get("#{origin}/robots.txt", accept: "text/plain")
|
|
112
|
+
return Robots.new(nil) if response.code.start_with?("4")
|
|
113
|
+
return "redirect" if response.is_a?(Net::HTTPRedirection)
|
|
114
|
+
return "robots_unreachable" unless response.is_a?(Net::HTTPSuccess)
|
|
115
|
+
|
|
116
|
+
Robots.new(response.body.to_s.dup.force_encoding(Encoding::UTF_8))
|
|
117
|
+
rescue StandardError
|
|
118
|
+
"robots_unreachable"
|
|
119
|
+
end
|
|
120
|
+
|
|
121
|
+
# Product types and tags repeat across a catalogue, and Classifier
|
|
122
|
+
# re-reads its YAML on every call.
|
|
123
|
+
def memoized_classifier
|
|
124
|
+
memo = {}
|
|
125
|
+
->(text) { memo[text] ||= Classifier.categories_for(text) }
|
|
126
|
+
end
|
|
127
|
+
|
|
128
|
+
def handoff_only?(origin)
|
|
129
|
+
@handoff_only.host?(URI.parse(origin.to_s).host)
|
|
130
|
+
rescue URI::InvalidURIError
|
|
131
|
+
true
|
|
132
|
+
end
|
|
133
|
+
|
|
134
|
+
def note(status, reason, pages, products)
|
|
135
|
+
{ "status" => status, "reason" => reason, "pages" => pages, "products" => products, "at" => @now.to_i }
|
|
136
|
+
end
|
|
137
|
+
|
|
138
|
+
def store_sighting(origin, crawl)
|
|
139
|
+
fields = { crawl: crawl }
|
|
140
|
+
fields[:platform] = "shopify" unless crawl["status"] == "skipped"
|
|
141
|
+
{ origin: origin, url: nil, title: nil, brand: nil, gtin: nil, store_fields: fields }
|
|
142
|
+
end
|
|
143
|
+
end
|
|
144
|
+
end
|
|
145
|
+
end
|
|
146
|
+
end
|
|
147
|
+
end
|
|
@@ -3,6 +3,7 @@ require_relative "sources/stores_file"
|
|
|
3
3
|
require_relative "sources/browser"
|
|
4
4
|
require_relative "sources/wikidata"
|
|
5
5
|
require_relative "sources/webmcp_sweep"
|
|
6
|
+
require_relative "sources/storefront_products"
|
|
6
7
|
|
|
7
8
|
module Portage
|
|
8
9
|
module Cli
|
|
@@ -16,7 +17,8 @@ module Portage
|
|
|
16
17
|
"stores_file" => -> { StoresFile.new },
|
|
17
18
|
"browser" => -> { Browser.new },
|
|
18
19
|
"wikidata" => -> { Wikidata.new },
|
|
19
|
-
"webmcp_sweep" => -> { WebmcpSweep.new }
|
|
20
|
+
"webmcp_sweep" => -> { WebmcpSweep.new },
|
|
21
|
+
"storefront_products" => -> { StorefrontProducts.new }
|
|
20
22
|
}.freeze
|
|
21
23
|
|
|
22
24
|
# Runs with no extra opt-in: no bridge required, and no low-yield
|
|
@@ -24,7 +26,8 @@ module Portage
|
|
|
24
26
|
# show up in `portage index sources`, just not run unless named in
|
|
25
27
|
# `--sources`. `browser` is listed but never a default — it yields
|
|
26
28
|
# nothing; `portage browser import` writes those entries itself
|
|
27
|
-
# (see Sources::Browser).
|
|
29
|
+
# (see Sources::Browser). `storefront_products` sends up to 21
|
|
30
|
+
# requests a store, so it runs only when named.
|
|
28
31
|
DEFAULT_NAMES = %w[shopify_catalog stores_file].freeze
|
|
29
32
|
|
|
30
33
|
def self.all = ALL.values.map(&:call)
|
|
@@ -1,12 +1,13 @@
|
|
|
1
|
-
require "json"
|
|
2
|
-
require "fileutils"
|
|
3
1
|
require "uri"
|
|
2
|
+
require_relative "database"
|
|
4
3
|
|
|
5
4
|
module Portage
|
|
6
5
|
module Cli
|
|
7
6
|
module Index
|
|
8
|
-
# `~/.portage/index/
|
|
9
|
-
# `
|
|
7
|
+
# The `stores` table of `~/.portage/index/index.sqlite3` (Index::Database;
|
|
8
|
+
# `stores.json` before docs/plans/local-catalogue.md Phase 1) — one
|
|
9
|
+
# entry per merchant origin `portage index build`/`refresh`/`add` has
|
|
10
|
+
# found and verified.
|
|
10
11
|
#
|
|
11
12
|
# This is untrusted, user-owned data, same posture as `stores.yml`
|
|
12
13
|
# before it's tagged: never checked into git (docs/plans/
|
|
@@ -19,7 +20,7 @@ module Portage
|
|
|
19
20
|
# search result (see search_backends_spec.rb/cli_spec.rb for the
|
|
20
21
|
# specs proving both).
|
|
21
22
|
#
|
|
22
|
-
# Entry shape (string keys, JSON
|
|
23
|
+
# Entry shape (string keys, a JSON object per row): origin, platform,
|
|
23
24
|
# ucp_version, capabilities (array of "catalog"/"cart"/"checkout"),
|
|
24
25
|
# webmcp_preset, categories (category id => weight, top 5),
|
|
25
26
|
# sources (array of source names that found it), last_verified
|
|
@@ -27,45 +28,49 @@ module Portage
|
|
|
27
28
|
class Store
|
|
28
29
|
PATH = File.join(Dir.home, ".portage", "index", "stores.json").freeze
|
|
29
30
|
|
|
31
|
+
# @param path [String] where the legacy stores.json lives (or would
|
|
32
|
+
# live) — the database sits beside it, and a stores.json found
|
|
33
|
+
# there is imported on first open.
|
|
30
34
|
def initialize(path: PATH)
|
|
31
|
-
@
|
|
35
|
+
@db = Database.new(path: Database.path_for(path))
|
|
32
36
|
end
|
|
33
37
|
|
|
34
|
-
def all = entries.values
|
|
38
|
+
def all = @db.entries("stores").values
|
|
35
39
|
|
|
36
|
-
def find(origin) =
|
|
40
|
+
def find(origin) = @db.get("stores", origin)
|
|
37
41
|
|
|
38
42
|
# Merges `fields` onto whatever's already there for `origin` (or
|
|
39
43
|
# starts a fresh entry) — so a second source finding the same store
|
|
40
44
|
# adds to its `sources`/`categories` rather than clobbering the
|
|
41
45
|
# first source's findings.
|
|
42
46
|
def upsert(origin, **fields)
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
+
@db.transaction do
|
|
48
|
+
existing = @db.get("stores", origin) || { "origin" => origin }
|
|
49
|
+
existing.merge(fields.transform_keys(&:to_s)).tap { |merged| @db.put("stores", origin, merged) }
|
|
50
|
+
end
|
|
47
51
|
end
|
|
48
52
|
|
|
49
53
|
# @return [Integer] how many entries (0 or 1 — origins are unique)
|
|
50
54
|
# this host's removal actually dropped.
|
|
51
55
|
def remove(host)
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
+
@db.transaction do
|
|
57
|
+
doomed = @db.entries("stores").keys.select { |origin| host_of(origin) == host }
|
|
58
|
+
doomed.each { |origin| @db.delete("stores", origin) }
|
|
59
|
+
doomed.length
|
|
60
|
+
end
|
|
56
61
|
end
|
|
57
62
|
|
|
58
63
|
# @return [Integer, nil] seconds since the oldest entry was
|
|
59
64
|
# verified — nil when the index is empty. `doctor` and `index
|
|
60
65
|
# show` use this to say how stale the local index is.
|
|
61
66
|
def oldest_verified_age(now: Time.now)
|
|
62
|
-
timestamps =
|
|
67
|
+
timestamps = all.filter_map { |e| e["last_verified"] }
|
|
63
68
|
return nil if timestamps.empty?
|
|
64
69
|
|
|
65
70
|
now.to_i - timestamps.min
|
|
66
71
|
end
|
|
67
72
|
|
|
68
|
-
def exists? =
|
|
73
|
+
def exists? = @db.exists?
|
|
69
74
|
|
|
70
75
|
private
|
|
71
76
|
|
|
@@ -74,35 +79,6 @@ module Portage
|
|
|
74
79
|
rescue URI::InvalidURIError
|
|
75
80
|
""
|
|
76
81
|
end
|
|
77
|
-
|
|
78
|
-
def entries
|
|
79
|
-
@entries ||= read
|
|
80
|
-
end
|
|
81
|
-
|
|
82
|
-
def read
|
|
83
|
-
return {} unless File.readable?(@path)
|
|
84
|
-
|
|
85
|
-
# A business name/title can carry non-ASCII bytes (curly quotes,
|
|
86
|
-
# ®, accents); reading with the process's default external
|
|
87
|
-
# encoding (US-ASCII on a bare-minimal LANG, confirmed live in a
|
|
88
|
-
# sandbox with no locale set) would otherwise raise on the very
|
|
89
|
-
# first non-ASCII byte and get silently swallowed below, quietly
|
|
90
|
-
# dropping the whole file's worth of entries — this was caught by
|
|
91
|
-
# this phase's own live index-build check.
|
|
92
|
-
parsed = JSON.parse(File.read(@path, encoding: "UTF-8"))
|
|
93
|
-
parsed.is_a?(Hash) ? parsed : {}
|
|
94
|
-
rescue StandardError
|
|
95
|
-
{}
|
|
96
|
-
end
|
|
97
|
-
|
|
98
|
-
# An index that can't be written just means this run's findings
|
|
99
|
-
# aren't saved — never a failed build.
|
|
100
|
-
def write
|
|
101
|
-
FileUtils.mkdir_p(File.dirname(@path))
|
|
102
|
-
File.write(@path, JSON.generate(@entries))
|
|
103
|
-
rescue StandardError
|
|
104
|
-
nil
|
|
105
|
-
end
|
|
106
82
|
end
|
|
107
83
|
end
|
|
108
84
|
end
|
data/lib/portage/cli/index.rb
CHANGED
|
@@ -49,6 +49,19 @@ module Portage
|
|
|
49
49
|
|
|
50
50
|
def self.retail_handoff_host?(host) = HandoffOnly.matches_any?(host, RETAIL_HANDOFF_HOSTS)
|
|
51
51
|
|
|
52
|
+
# Adds the UCP Product wire hash to an offer as `product` (docs/plans/
|
|
53
|
+
# local-catalogue.md Phase 3), as the store served it, with `media`
|
|
54
|
+
# cut to the first image. Shared by Find and ShopifyCatalog so both
|
|
55
|
+
# trim the same way. Anything that isn't a non-empty hash leaves the
|
|
56
|
+
# offer as it was: the field is left out, not faked.
|
|
57
|
+
def self.with_product(offer, product)
|
|
58
|
+
return offer unless product.is_a?(Hash) && !product.empty?
|
|
59
|
+
|
|
60
|
+
media = product["media"]
|
|
61
|
+
product = product.merge("media" => media.first(1)) if media.is_a?(Array) && media.length > 1
|
|
62
|
+
offer.merge(product: product)
|
|
63
|
+
end
|
|
64
|
+
|
|
52
65
|
# Shared by every retailer source below: an offer's `store` is the
|
|
53
66
|
# origin of the item URL the API itself returned, same idea as
|
|
54
67
|
# ShopifyCatalog#origin_of — kept as one module method instead of
|
|
@@ -133,8 +146,10 @@ module Portage
|
|
|
133
146
|
return nil unless origin
|
|
134
147
|
|
|
135
148
|
amount, currency = price_of(product)
|
|
136
|
-
|
|
137
|
-
|
|
149
|
+
OfferSources.with_product(
|
|
150
|
+
{ store: origin, source: name, checkout: nil, product_id: variant["id"], title: product["title"],
|
|
151
|
+
amount: amount, currency: currency, url: variant["url"] }, product
|
|
152
|
+
)
|
|
138
153
|
end
|
|
139
154
|
|
|
140
155
|
def origin_of(url)
|
data/lib/portage/cli/version.rb
CHANGED