portage-cli 0.10.0 → 0.11.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,40 @@
1
+ # Words Classifier never uses as evidence of a category, on either side: script/categories
2
+ # leaves them out of categories.yml, and Classifier.categories_for drops them from its
3
+ # input. Each entry is `word: why`. Both forms of a word are listed, because the stoplist is
4
+ # matched exactly (the keyword match is what normalizes plurals).
5
+ #
6
+ # Two kinds of word belong here: words that describe how something is sold or filed rather
7
+ # than what it is, and words that label a whole family of nodes ("accessories" names 76 of
8
+ # the 192 level-2 nodes). A specific word that is merely ambiguous does not belong here.
9
+ and: function word; appears in taxonomy names ("Hardware & Tools")
10
+ for: function word; appears in taxonomy names ("Filters for ...")
11
+ the: function word
12
+ with: function word ("Vac with Multi-Purpose Head")
13
+ from: function word
14
+ new: merchandising flag; Light Yard tags products "New Collection"
15
+ sale: merchandising flag; a sale tag says nothing about the product
16
+ gift: merchandising flag; "Gift Set" is a way to sell, not a product type
17
+ gifts: merchandising flag; plural of gift
18
+ set: bundle word; "Gift Set" and "Furniture Set" say how it is packaged
19
+ sets: bundle word; plural of set
20
+ kit: bundle word; "drill kit" is the drill
21
+ kits: bundle word; plural of kit
22
+ collection: storefront and url word; matched "Toll Collection Devices" (4488) for every "New Collection" tag
23
+ collections: url and storefront word; the /collections/ path segment of every Shopify collection url
24
+ products: url word; the /products/ path segment of every Shopify product url
25
+ product: url and page-title word
26
+ category: url word; the /category/ path segment
27
+ shop: page-title word ("Cordless Drills | Hardware | Shop")
28
+ accessories: names 76 of the 192 level-2 nodes, so it cannot tell them apart
29
+ accessory: singular of accessories
30
+ supplies: names 25 level-2 nodes ("Pet Supplies", "Office Supplies"), so it cannot tell them apart
31
+ supply: singular of supplies
32
+ equipment: names 21 level-2 nodes, so it cannot tell them apart
33
+ parts: names 21 level-2 nodes ("Vehicle Parts"), so it cannot tell them apart
34
+ replacement: a spare, not a product type; "Replacement Fade Blade Set"
35
+ general: filler word ("General Office Supplies")
36
+ other: filler word; the catch-all name in many taxonomies
37
+ miscellaneous: catch-all product type; JB Hi-Fi files 237 of its first 1,250 products under "MISCELLANEOUS"
38
+ service: generic product-type word; JB Hi-Fi's "TELCO SERVICES" type matched Food Service (135)
39
+ services: plural of service; JB Hi-Fi files 279 of its first 1,250 products under "TELCO SERVICES"
40
+ hand: craft marker; Light Yard tags 146 of its 164 products "British Hand-Made", and "hand" also names 10 level-2 nodes ("Hand Tools")
@@ -0,0 +1,15 @@
1
+ # Words the Google taxonomy has no node for, added to a node's keywords by script/categories.
2
+ # `id:` then `word: why`; the why names the golden case (spec/fixtures/classifier_golden.yml)
3
+ # that needs it, by its text. Add a word here only for a case the golden set proves, never
4
+ # in bulk.
5
+ '594':
6
+ pendant: pendant light
7
+ sconce: wall sconce
8
+ bollard: Bollard light Bollard Lights British Hand-Made Driveway Lights Path Lights
9
+ Patio Lights Wooden Lights £250-£500
10
+ '604':
11
+ whitegoods: WHITEGOODS Brand:Beko LimitedStock
12
+ '262':
13
+ telco: TELCO SERVICES Brand:Samsung InStock
14
+ '187':
15
+ boots: leather boots
@@ -21,7 +21,8 @@ module Portage
21
21
  PRODUCT_PATH = %r{/products?/[^/?#]+}
22
22
 
23
23
  # @param rows [Array<Hash>] Readers rows, all for one domain.
24
- # @return [Hash{String => Integer}] category id => weight, top 5.
24
+ # @return [Hash{String => Integer}] category id => weight, top 5 (equal
25
+ # weights in the order first seen).
25
26
  def self.domain(rows)
26
27
  texts = Hash.new(0)
27
28
  rows.each { |row| texts[text_of(row)] += row[:visits] }
@@ -29,7 +30,9 @@ module Portage
29
30
  texts.each do |text, visits|
30
31
  Classifier.categories_for(text).each { |id| tally[id] += visits } unless text.empty?
31
32
  end
32
- tally.sort_by { |_id, weight| -weight }.first(TOP_CATEGORIES).to_h
33
+ # `sort_by` isn't stable, so the position breaks ties: first seen wins.
34
+ tally.each_with_index.sort_by { |(_id, weight), seen| [-weight, seen] }.first(TOP_CATEGORIES)
35
+ .to_h { |pair, _seen| pair }
33
36
  end
34
37
 
35
38
  # @param kept [Array<Hash>] Importer's kept entries, rows included.
@@ -57,11 +57,21 @@ module Portage
57
57
  report = report.merge(adapter: adapter_for(report))
58
58
  report = report.merge(webmcp: webmcp_for(report))
59
59
  verdict = verdict_for(report)
60
- report.merge(verdict: verdict, next_step: CheckNextStep.call(verdict, report))
60
+ with_index_hint(report.merge(verdict: verdict, next_step: CheckNextStep.call(verdict, report)))
61
61
  end
62
62
 
63
63
  private
64
64
 
65
+ # docs/plans/local-catalogue.md Phase 2: a Shopify (or native UCP)
66
+ # store's catalogue can be crawled into the local index. Check only
67
+ # names the command; it never crawls.
68
+ def with_index_hint(report)
69
+ return report unless report[:native_ucp] || report[:platform] == "Shopify"
70
+
71
+ port = @uri.port == @uri.default_port ? "" : ":#{@uri.port}"
72
+ report.merge(index_hint: "portage index add #{@uri.scheme}://#{@uri.host}#{port} --crawl")
73
+ end
74
+
65
75
  def handoff_only_report
66
76
  { url: @uri.to_s, native_ucp: nil, platform: nil, recommended_gem: nil, handoff_only: true, adapter: nil,
67
77
  webmcp: webmcp_skipped("#{@uri.host} is hand-off only, so it isn't probed"),
@@ -0,0 +1,135 @@
1
+ module Portage
2
+ module Cli
3
+ module Classifier
4
+ # Scores the tokenized input against every node and keeps the best few
5
+ # (docs/plans/local-catalogue.md, Phase 5). Nodes carry hundreds of
6
+ # rolled-up keywords, so this is where "a generic word matches a dozen
7
+ # categories" gets dealt with.
8
+ module Ranking
9
+ # A node's own name words and its descendants' (`keywords`) say what
10
+ # the product is; its parent's (`parent_keywords`, e.g. "home",
11
+ # "garden") only where it is filed, so they count half as much.
12
+ KEYWORD_WEIGHT = 2
13
+ PARENT_WEIGHT = 1
14
+
15
+ # Callers treat every returned id as evidence (store routing, a
16
+ # browser domain's category tally), so the answer keeps only the best
17
+ # MAX_CATEGORIES ids and, of those, only ones scoring at least
18
+ # 1/CUTOFF of the best: "electric kettle" is Kitchen & Dining, not
19
+ # also Chairs because of "electric".
20
+ MAX_CATEGORIES = 3
21
+ CUTOFF = 2
22
+
23
+ module_function
24
+
25
+ # @param counts [Hash{String => Integer}] word => times the input says it.
26
+ # @param table [Classifier::Table]
27
+ # @param stopped [Hash] the stoplist.
28
+ # @return [Array<String>] up to MAX_CATEGORIES node ids, best first.
29
+ def best(counts, table, stopped)
30
+ ranked = score(counts, table).map do |id, (score, evidence)|
31
+ [id, score, evidence, *tie_breakers(table.nodes[id], counts.keys, stopped)]
32
+ end
33
+ ranked = ranked.sort_by { |(_id, score, _evidence, *rest)| [-score.round(6), *rest] }
34
+ cut(ranked).first(MAX_CATEGORIES).map(&:first)
35
+ end
36
+
37
+ # The first row always stays; later rows only if strong?.
38
+ def cut(ranked)
39
+ strongest = ranked.map { |row| row[2] }.max
40
+ ranked.each_with_index.select { |row, i| i.zero? || strong?(row[2], strongest) }.map(&:first)
41
+ end
42
+
43
+ # Distinct input words and how often each occurs. A plural variant of
44
+ # a word counts as the same word ("pendant" and "Pendants").
45
+ def word_counts(tokens)
46
+ tokens.each_with_object({}) do |token, counts|
47
+ word = counts.keys.find { |known| Classifier.word_match?(token, known) } || token
48
+ counts[word] = counts.fetch(word, 0) + 1
49
+ end
50
+ end
51
+
52
+ # @return [Hash{String => Array(Float, Float)}] node id => [score,
53
+ # evidence], for nodes scoring above zero. Each word counts once for
54
+ # a node, however many of its keywords match ("light" and "lights"
55
+ # are two keywords but one word of the input), so a node cannot win
56
+ # by listing a word's variants: KEYWORD_WEIGHT when a `keywords`
57
+ # entry matches, PARENT_WEIGHT when only a `parent_keywords` entry
58
+ # does, times how often the input says the word (1 + ln count: a
59
+ # Shopify tag list repeats "Pendant Lights" once per room, which is
60
+ # its best evidence, and the log keeps a long tag list from burying
61
+ # a different word). `evidence` is the same sum with each word also
62
+ # weighted by how rare it is (ln(1 + nodes / nodes it matches)), so
63
+ # "kettle" counts for more than "electric".
64
+ def score(counts, table)
65
+ scores = Hash.new { |hash, id| hash[id] = [0.0, 0.0] }
66
+ counts.each do |word, count|
67
+ contributions(word, count, table).each { |id, weight, rarity| add(scores[id], weight, rarity) }
68
+ end
69
+ scores
70
+ end
71
+
72
+ # @return [Array<Array(String, Float, Float)>] [node id, weight,
73
+ # rarity] for every node `word` matches.
74
+ def contributions(word, count, table)
75
+ own = matching_ids(word, table.own)
76
+ parent = matching_ids(word, table.parent) - own
77
+ return [] if own.empty? && parent.empty?
78
+
79
+ tf = 1 + Math.log(count)
80
+ rarity = Math.log(1 + table.nodes.size.fdiv((own + parent).length))
81
+ own.map { |id| [id, KEYWORD_WEIGHT * tf, rarity] } + parent.map { |id| [id, PARENT_WEIGHT * tf, rarity] }
82
+ end
83
+
84
+ def add(pair, weight, rarity)
85
+ pair[0] += weight
86
+ pair[1] += weight * rarity
87
+ end
88
+
89
+ # Ranking by `score` alone is what classifies a long tag list best,
90
+ # because rare words are mostly noise there (a room name in a
91
+ # lighting store's tags). The cut uses `evidence`, which a generic
92
+ # word cannot fill: an id beyond the first stays only if its evidence
93
+ # reaches 1/CUTOFF of the strongest.
94
+ def strong?(evidence, strongest) = (evidence * CUTOFF) - strongest > -1e-9
95
+
96
+ # What separates nodes with equal scores, best first: the share of
97
+ # the node's own name the input covers ("Sofas" before "Sofa
98
+ # Accessories" for "sofa"), then a name with no stoplisted word in it
99
+ # ("Household Appliances" before "Household Appliance Accessories"),
100
+ # then fewer keywords (the smaller node is the more specific one),
101
+ # then the file's own order.
102
+ def tie_breakers(node, words, stopped)
103
+ name = node["name"].to_s.split(" > ").last.to_s.downcase.split(/[^\p{Alpha}]+/)
104
+ .select { |word| word.length >= Classifier::MIN_WORD_LENGTH }
105
+ covered = name.count { |part| words.any? { |word| Classifier.word_match?(word, part) } }
106
+ generic = name.any? { |part| stopped.key?(part) } ? 1 : 0
107
+ [-covered.fdiv([name.length, 1].max), generic, Array(node["keywords"]).length, node["order"]]
108
+ end
109
+
110
+ # @return [Array<String>] ids of the nodes in `index` with a keyword
111
+ # `word` matches.
112
+ def matching_ids(word, index)
113
+ keywords_matching(word).flat_map { |keyword| index.fetch(keyword, []) }.uniq
114
+ end
115
+
116
+ # The keywords Classifier.word_match? accepts `word` for: itself, the
117
+ # singular it is a plural of ("boots" -> "boot", "watches" ->
118
+ # "watch"), the plural of it ("boot" -> "boots"), and the "y"/"ies"
119
+ # swap. Looking a word's few candidate keywords up in a hash, rather
120
+ # than comparing every word with every keyword, keeps a long
121
+ # product-tag text (Index::Sources::StorefrontProducts,
122
+ # docs/plans/local-catalogue.md Phase 2) cheap now that nodes carry
123
+ # hundreds of keywords.
124
+ def keywords_matching(word)
125
+ keywords = [word, "#{word}s", "#{word}es"]
126
+ keywords << word.delete_suffix("s") if word.end_with?("s")
127
+ keywords << word.delete_suffix("es") if word.end_with?("es")
128
+ keywords << "#{word[0..-4]}y" if word.end_with?("ies")
129
+ keywords << "#{word[0..-2]}ies" if word.end_with?("y")
130
+ keywords
131
+ end
132
+ end
133
+ end
134
+ end
135
+ end
@@ -0,0 +1,63 @@
1
+ require "yaml"
2
+
3
+ module Portage
4
+ module Cli
5
+ module Classifier
6
+ # The shipped categories.yml and the user's ~/.portage/categories.yml,
7
+ # merged (a user node replaces a shipped node of the same id), with the
8
+ # keyword => node ids indexes the scoring looks words up in.
9
+ Table = Struct.new(:nodes, :own, :parent)
10
+
11
+ class Table
12
+ @cache = {}
13
+
14
+ class << self
15
+ # Built once per pair of files and kept for the process, keyed by
16
+ # the files' mtime and size, so an edit to ~/.portage/categories.yml
17
+ # still shows up on the next call; classifying a whole catalogue
18
+ # would otherwise rebuild the index for every product.
19
+ def for(known_path, user_path)
20
+ key = [signature(known_path), signature(user_path)]
21
+ @cache.clear if @cache.size > 8
22
+ @cache[key] ||= build(known_path, user_path)
23
+ end
24
+
25
+ # @return [Hash] the YAML mapping at `path`; empty when the file is
26
+ # missing, unreadable or not a mapping.
27
+ def load_yaml(path)
28
+ return {} unless path && File.readable?(path)
29
+
30
+ data = YAML.safe_load_file(path)
31
+ data.is_a?(Hash) ? data : {}
32
+ rescue StandardError
33
+ {}
34
+ end
35
+
36
+ private
37
+
38
+ def build(known_path, user_path)
39
+ ordered = {}
40
+ load_yaml(known_path).each_with_index { |(id, node), i| ordered[id] = node.merge("order" => i) }
41
+ load_yaml(user_path).each_with_index do |(id, node), i|
42
+ ordered[id] = node.merge("order" => ordered.size + i)
43
+ end
44
+ new(ordered, index(ordered, "keywords"), index(ordered, "parent_keywords"))
45
+ end
46
+
47
+ def index(nodes, field)
48
+ result = Hash.new { |hash, key| hash[key] = [] }
49
+ nodes.each { |id, node| Array(node[field]).each { |keyword| result[keyword] << id } }
50
+ result.default_proc = nil
51
+ result
52
+ end
53
+
54
+ def signature(path)
55
+ return nil unless path && File.readable?(path)
56
+
57
+ [path, File.mtime(path).to_r, File.size(path)]
58
+ end
59
+ end
60
+ end
61
+ end
62
+ end
63
+ end
@@ -1,4 +1,6 @@
1
1
  require "yaml"
2
+ require_relative "classifier/table"
3
+ require_relative "classifier/ranking"
2
4
 
3
5
  module Portage
4
6
  module Cli
@@ -18,10 +20,21 @@ module Portage
18
20
  # check matches in both directions ("carpet" contains "pet", "chair"
19
21
  # contains "hair", "scarf" contains "car") and was a real source of
20
22
  # false positives before this became whole-word.
23
+ #
24
+ # `known-stores/categories.yml` is generated by `script/categories`: a level-2
25
+ # node's `keywords` are its own name plus its descendants', `parent_keywords`
26
+ # are its parent's words, and `category-stoplist.yml` words are left out.
27
+ # See Ranking for how a text is scored against it.
21
28
  module Classifier
22
29
  # `known-stores/categories.yml`, from `lib/portage/cli/classifier.rb`.
23
30
  KNOWN_PATH = File.expand_path("../../../known-stores/categories.yml", __dir__).freeze
24
31
 
32
+ # Words that are never evidence of a category (`known-stores/category-stoplist.yml`,
33
+ # `word: why`): merchandising and url words, and words that name a whole family of
34
+ # nodes. `script/categories` leaves them out of the keywords; `categories_for`
35
+ # drops them from its input, so the two sides agree.
36
+ STOPLIST_PATH = File.expand_path("../../../known-stores/category-stoplist.yml", __dir__).freeze
37
+
25
38
  # The user's own additions/overrides — same id overrides a shipped
26
39
  # node's keywords, a new id extends the taxonomy. Absent by default;
27
40
  # nothing here is required for the shipped file to work.
@@ -44,21 +57,22 @@ module Portage
44
57
  # @param known_path [String] override for KNOWN_PATH — specs redirect
45
58
  # this the same way SearchBackends::Allowlist takes its own `path:`.
46
59
  # @param user_path [String] override for PATH.
47
- # @return [Array<String>] category ids, most keyword hits first. Ties
48
- # keep the shipped file's own order. Empty when nothing matches.
49
- def self.categories_for(text, known_path: KNOWN_PATH, user_path: PATH)
50
- words = tokenize(text)
51
- return [] if words.empty?
52
-
53
- scored = nodes(known_path, user_path).filter_map { |id, node| rank(id, node, words) }
54
- scored.sort_by { |(_id, score, order)| [-score, order] }.map(&:first)
60
+ # @param stoplist_path [String] override for STOPLIST_PATH.
61
+ # @return [Array<String>] up to MAX_CATEGORIES category ids, best score
62
+ # first (ties: see .tie_breakers). Empty when nothing matches.
63
+ def self.categories_for(text, known_path: KNOWN_PATH, user_path: PATH, stoplist_path: STOPLIST_PATH)
64
+ stopped = Table.load_yaml(stoplist_path)
65
+ counts = Ranking.word_counts(tokenize(text).reject { |word| stopped.key?(word) })
66
+ return [] if counts.empty?
67
+
68
+ Ranking.best(counts, Table.for(known_path, user_path), stopped)
55
69
  end
56
70
 
57
71
  # @return [Array<String>] the taxonomy names for `ids`, in order —
58
72
  # `portage browser import` (Phase 3) shows a domain's guessed
59
73
  # categories by name so the user can judge them before saving.
60
74
  def self.names_for(ids, known_path: KNOWN_PATH, user_path: PATH)
61
- all = nodes(known_path, user_path)
75
+ all = Table.for(known_path, user_path).nodes
62
76
  Array(ids).filter_map { |id| all.dig(id.to_s, "name") }
63
77
  end
64
78
 
@@ -88,17 +102,6 @@ module Portage
88
102
  .select { |word| word.length >= MIN_WORD_LENGTH }
89
103
  end
90
104
 
91
- # --- Scoring one node against the tokenized input ---
92
-
93
- def self.rank(id, node, words)
94
- keywords = Array(node["keywords"])
95
- hits = keywords.count { |keyword| words.any? { |word| word_match?(word, keyword) } }
96
- return nil unless hits.positive?
97
-
98
- [id, hits, node["order"].to_i]
99
- end
100
- private_class_method :rank
101
-
102
105
  # Whole-word only, plus the plural forms a keyword list and a real
103
106
  # query/title actually differ by: an exact match, one plus a trailing
104
107
  # "s" or "es" ("boot"/"boots", "watch"/"watches"), or the "y"/"ies"
@@ -129,30 +132,6 @@ module Portage
129
132
  (keyword.end_with?("ies") && word == "#{keyword[0..-4]}y")
130
133
  end
131
134
  private_class_method :ies_y_match?
132
-
133
- # --- Loading and merging the two files ---
134
-
135
- # Re-read on every call rather than cached process-wide: `find` calls
136
- # this once or twice per invocation, not in a hot loop, and a cached
137
- # copy would miss an edit to ~/.portage/categories.yml until the next
138
- # process start.
139
- def self.nodes(known_path, user_path)
140
- ordered = {}
141
- load_yaml(known_path).each_with_index { |(id, node), i| ordered[id] = node.merge("order" => i) }
142
- load_yaml(user_path).each_with_index { |(id, node), i| ordered[id] = node.merge("order" => ordered.size + i) }
143
- ordered
144
- end
145
- private_class_method :nodes
146
-
147
- def self.load_yaml(path)
148
- return {} unless path && File.readable?(path)
149
-
150
- data = YAML.safe_load_file(path)
151
- data.is_a?(Hash) ? data : {}
152
- rescue StandardError
153
- {}
154
- end
155
- private_class_method :load_yaml
156
135
  end
157
136
  end
158
137
  end
@@ -42,7 +42,8 @@ module Portage
42
42
  # those four warnings were noise on every fresh install.
43
43
  def initialize(adapter_class: nil, proxy_settings: ProxySettings.new, install_doctor: InstallDoctor.new,
44
44
  seller: true, dot_env_path: DotEnv.loaded_path, index_stores: Index::Store.new,
45
- index_products: Index::ProductStore.new, known_cache: Index::KnownCache.new)
45
+ index_products: Index::ProductStore.new, known_cache: Index::KnownCache.new,
46
+ index_database: Index::Database.new(path: Index::Database.path_for(Index::Store::PATH)))
46
47
  @adapter_class = adapter_class
47
48
  @proxy_settings = proxy_settings
48
49
  @install_doctor = install_doctor
@@ -51,6 +52,7 @@ module Portage
51
52
  @index_stores = index_stores
52
53
  @index_products = index_products
53
54
  @known_cache = known_cache
55
+ @index_database = index_database
54
56
  end
55
57
 
56
58
  # `portage setup` always offers its wizard on a TTY; a bare `portage
@@ -210,12 +212,20 @@ module Portage
210
212
  # leaves behind.
211
213
  def index_finding
212
214
  refresh_known_cache_if_stale
213
- return Finding.new(check: "index", level: "info", message: "#{index_message}\n#{known_cache_message}") \
214
- if @index_stores.exists?
215
+ first = @index_stores.exists? ? index_message : no_index_message
216
+ database = @index_database.info
217
+ Finding.new(check: "index", level: "info", details: { database: database },
218
+ message: [first, database_message(database), known_cache_message].join("\n"))
219
+ end
220
+
221
+ def no_index_message
222
+ "No local index yet — run `portage index build` to give `find` a list of " \
223
+ "stores/products on top of stores.yml and web search."
224
+ end
215
225
 
216
- Finding.new(check: "index", level: "info",
217
- message: "No local index yet — run `portage index build` to give `find` a list of " \
218
- "stores/products on top of stores.yml and web search.\n#{known_cache_message}")
226
+ def database_message(info)
227
+ "Index database: #{info[:path]} (#{info[:stores]} store row(s), #{info[:products]} product row(s), " \
228
+ "FTS5 #{info[:fts5] ? 'available' : 'not available'})."
219
229
  end
220
230
 
221
231
  def refresh_known_cache_if_stale
@@ -229,9 +229,11 @@ module Portage
229
229
  amount, currency = price_of(product)
230
230
  return nil if over_max_price?(amount)
231
231
 
232
- { store: store[:origin], source: store[:source], checkout: store[:checkout],
233
- product_id: field(product, "id"), title: field(product, "title"),
234
- amount: amount, currency: currency, url: field(product, "url") }
232
+ OfferSources.with_product(
233
+ { store: store[:origin], source: store[:source], checkout: store[:checkout],
234
+ product_id: field(product, "id"), title: field(product, "title"),
235
+ amount: amount, currency: currency, url: field(product, "url") }, product
236
+ )
235
237
  end
236
238
 
237
239
  # An unpriced offer stays in: no price isn't the same as too dear.
@@ -37,6 +37,9 @@ module Portage
37
37
  THROTTLE = 0.1
38
38
  STALE_AFTER = 7 * 24 * 60 * 60
39
39
  TOP_CATEGORIES = 5
40
+ # Product sightings per ProductStore#upsert_many transaction — one
41
+ # products.json page's worth (docs/plans/local-catalogue.md Phase 2).
42
+ WRITE_BATCH = 250
40
43
 
41
44
  # Public so BrowserImport::Importer (Phase 3) labels a probed
42
45
  # origin's capabilities exactly the way an index build does.
@@ -94,13 +97,19 @@ module Portage
94
97
  # HandoffOnly) is recorded without ever probing it — the user
95
98
  # explicitly named it, but that's still not a request this process
96
99
  # sends.
97
- def add(url)
100
+ #
101
+ # `crawl: true` (`index add URL --crawl`) then reads the store's own
102
+ # catalogue through Sources::StorefrontProducts into the index.
103
+ # Opt-in: a crawl is up to 21 more requests and 20s of pauses, where
104
+ # a plain add is one probe.
105
+ def add(url, crawl: false)
98
106
  origin = origin_of(url)
99
107
  return { added: false, message: "Not a valid http(s) URL: #{url}" } unless origin
100
108
  return store_manual_handoff_only(origin) if handoff_only_origin?(origin)
101
109
 
102
110
  session = probe(origin)
103
- store_manual(origin, session)
111
+ result = store_manual(origin, session)
112
+ crawl ? crawl_added(origin, result) : result
104
113
  end
105
114
 
106
115
  # `portage index remove HOST`
@@ -184,10 +193,29 @@ module Portage
184
193
  { added: true, origin: origin, message: "Added #{origin} — hand-off only, never probed." }
185
194
  end
186
195
 
196
+ # `store_fields:` on a sighting (StorefrontProducts' crawl note and
197
+ # platform) lands on the store row as-is.
187
198
  def update_existing(origin, group)
188
199
  existing = @stores.find(origin)
200
+ fields = group.filter_map { |g| g[:store_fields] }.reduce({}, :merge)
189
201
  @stores.upsert(origin, sources: merged_sources(existing, group),
190
- categories: merge_categories(existing["categories"], group))
202
+ categories: merge_categories(existing["categories"], group), **fields)
203
+ end
204
+
205
+ def crawl_added(origin, result)
206
+ sightings = Sources::StorefrontProducts.new(stores: @stores, handoff_only: @handoff_only)
207
+ .crawl(origin, platform: @stores.find(origin)&.dig("platform"))
208
+ tagged = sightings.map { |s| s.merge(source: "storefront_products") }
209
+ apply(tagged, dry_run: false)
210
+ note = tagged.last.dig(:store_fields, :crawl)
211
+ result.merge(crawl: note, message: "#{result[:message]} #{crawl_message(note)}")
212
+ end
213
+
214
+ def crawl_message(note)
215
+ return "Catalogue not crawled (#{note['reason']})." if note["status"] == "skipped"
216
+
217
+ "Crawled #{note['products']} product(s) from #{note['pages']} page(s)" \
218
+ "#{" (stopped: #{note['reason']})" if note['reason']}."
191
219
  end
192
220
 
193
221
  def store_new(origin, session, group)
@@ -220,7 +248,7 @@ module Portage
220
248
  group.each do |sighting|
221
249
  next unless sighting[:title]
222
250
 
223
- Classifier.categories_for(sighting[:title]).each { |id| tally[id] += 1 }
251
+ categories_of(sighting).each { |id| tally[id] += 1 }
224
252
  end
225
253
  tally.sort_by { |_id, weight| -weight }.first(TOP_CATEGORIES).to_h
226
254
  end
@@ -232,17 +260,26 @@ module Portage
232
260
  def capabilities_of(session) = self.class.capabilities_of(session)
233
261
 
234
262
  def store_products(sightings)
235
- eligible = sightings.select { |s| s[:title] && @stores.find(s[:origin]) }
236
- eligible.each { |sighting| store_product(sighting) }
263
+ known = Hash.new { |memo, origin| memo[origin] = !@stores.find(origin).nil? }
264
+ eligible = sightings.select { |s| s[:title] && known[s[:origin]] }
265
+ eligible.each_slice(WRITE_BATCH) { |batch| @products.upsert_many(batch.map { |s| product_row(s) }) }
237
266
  eligible.length
238
267
  end
239
268
 
240
- def store_product(sighting)
241
- key = product_key(sighting)
242
- category = Classifier.categories_for(sighting[:title]).first
243
- @products.upsert(key, origin: sighting[:origin], seen_at: @now.to_i, title: sighting[:title],
244
- brand: sighting[:brand], gtin: sighting[:gtin], category: category,
245
- sources: [sighting[:source]].compact)
269
+ # `product:` on a sighting (StorefrontProducts' handle, url,
270
+ # image_url, options, variant_ids) is stored alongside the usual
271
+ # fields.
272
+ def product_row(sighting)
273
+ { key: product_key(sighting), origin: sighting[:origin], seen_at: @now.to_i, title: sighting[:title],
274
+ brand: sighting[:brand], gtin: sighting[:gtin], category: categories_of(sighting).first,
275
+ sources: [sighting[:source]].compact, **sighting.fetch(:product, {}) }
276
+ end
277
+
278
+ # A source that already classified its sighting (StorefrontProducts,
279
+ # on product_type and tags) says so in `categories:`; otherwise the
280
+ # title is classified here.
281
+ def categories_of(sighting)
282
+ sighting[:categories] || Classifier.categories_for(sighting[:title])
246
283
  end
247
284
 
248
285
  # GTIN when a source has one (none do yet); otherwise a normalized