portage-cli 0.10.0 → 0.11.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +52 -0
- data/README.md +20 -4
- data/known-stores/categories.yml +6841 -18
- data/known-stores/category-stoplist.yml +40 -0
- data/known-stores/category-synonyms.yml +15 -0
- data/lib/portage/cli/browser_import/categorize.rb +5 -2
- data/lib/portage/cli/check.rb +11 -1
- data/lib/portage/cli/classifier/ranking.rb +135 -0
- data/lib/portage/cli/classifier/table.rb +63 -0
- data/lib/portage/cli/classifier.rb +23 -44
- data/lib/portage/cli/doctor.rb +16 -6
- data/lib/portage/cli/find.rb +5 -3
- data/lib/portage/cli/index/builder.rb +49 -12
- data/lib/portage/cli/index/database.rb +150 -0
- data/lib/portage/cli/index/entry_product.rb +37 -0
- data/lib/portage/cli/index/legacy_import.rb +54 -0
- data/lib/portage/cli/index/product_store.rb +73 -35
- data/lib/portage/cli/index/schema.rb +70 -0
- data/lib/portage/cli/index/search.rb +73 -0
- data/lib/portage/cli/index/sources/storefront_products/mapper.rb +127 -0
- data/lib/portage/cli/index/sources/storefront_products/pages.rb +114 -0
- data/lib/portage/cli/index/sources/storefront_products/robots.rb +70 -0
- data/lib/portage/cli/index/sources/storefront_products.rb +147 -0
- data/lib/portage/cli/index/sources.rb +5 -2
- data/lib/portage/cli/index/store.rb +23 -47
- data/lib/portage/cli/index.rb +1 -0
- data/lib/portage/cli/offer_sources.rb +17 -2
- data/lib/portage/cli/version.rb +1 -1
- data/lib/portage/cli.rb +91 -10
- metadata +29 -2
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
# Words Classifier never uses as evidence of a category, on either side: script/categories
|
|
2
|
+
# leaves them out of categories.yml, and Classifier.categories_for drops them from its
|
|
3
|
+
# input. Each entry is `word: why`. Both forms of a word are listed, because the stoplist is
|
|
4
|
+
# matched exactly (the keyword match is what normalizes plurals).
|
|
5
|
+
#
|
|
6
|
+
# Two kinds of word belong here: words that describe how something is sold or filed rather
|
|
7
|
+
# than what it is, and words that label a whole family of nodes ("accessories" names 76 of
|
|
8
|
+
# the 192 level-2 nodes). A specific word that is merely ambiguous does not belong here.
|
|
9
|
+
and: function word; appears in taxonomy names ("Hardware & Tools")
|
|
10
|
+
for: function word; appears in taxonomy names ("Filters for ...")
|
|
11
|
+
the: function word
|
|
12
|
+
with: function word ("Vac with Multi-Purpose Head")
|
|
13
|
+
from: function word
|
|
14
|
+
new: merchandising flag; Light Yard tags products "New Collection"
|
|
15
|
+
sale: merchandising flag; a sale tag says nothing about the product
|
|
16
|
+
gift: merchandising flag; "Gift Set" is a way to sell, not a product type
|
|
17
|
+
gifts: merchandising flag; plural of gift
|
|
18
|
+
set: bundle word; "Gift Set" and "Furniture Set" say how it is packaged
|
|
19
|
+
sets: bundle word; plural of set
|
|
20
|
+
kit: bundle word; "drill kit" is the drill
|
|
21
|
+
kits: bundle word; plural of kit
|
|
22
|
+
collection: storefront and url word; matched "Toll Collection Devices" (4488) for every "New Collection" tag
|
|
23
|
+
collections: url and storefront word; the /collections/ path segment of every Shopify collection url
|
|
24
|
+
products: url word; the /products/ path segment of every Shopify product url
|
|
25
|
+
product: url and page-title word
|
|
26
|
+
category: url word; the /category/ path segment
|
|
27
|
+
shop: page-title word ("Cordless Drills | Hardware | Shop")
|
|
28
|
+
accessories: names 76 of the 192 level-2 nodes, so it cannot tell them apart
|
|
29
|
+
accessory: singular of accessories
|
|
30
|
+
supplies: names 25 level-2 nodes ("Pet Supplies", "Office Supplies"), so it cannot tell them apart
|
|
31
|
+
supply: singular of supplies
|
|
32
|
+
equipment: names 21 level-2 nodes, so it cannot tell them apart
|
|
33
|
+
parts: names 21 level-2 nodes ("Vehicle Parts"), so it cannot tell them apart
|
|
34
|
+
replacement: a spare, not a product type; "Replacement Fade Blade Set"
|
|
35
|
+
general: filler word ("General Office Supplies")
|
|
36
|
+
other: filler word; the catch-all name in many taxonomies
|
|
37
|
+
miscellaneous: catch-all product type; JB Hi-Fi files 237 of its first 1,250 products under "MISCELLANEOUS"
|
|
38
|
+
service: generic product-type word; JB Hi-Fi's "TELCO SERVICES" type matched Food Service (135)
|
|
39
|
+
services: plural of service; JB Hi-Fi files 279 of its first 1,250 products under "TELCO SERVICES"
|
|
40
|
+
hand: craft marker; Light Yard tags 146 of its 164 products "British Hand-Made", and "hand" also names 10 level-2 nodes ("Hand Tools")
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
# Words the Google taxonomy has no node for, added to a node's keywords by script/categories.
|
|
2
|
+
# `id:` then `word: why`; the why names the golden case (spec/fixtures/classifier_golden.yml)
|
|
3
|
+
# that needs it, by its text. Add a word here only for a case the golden set proves, never
|
|
4
|
+
# in bulk.
|
|
5
|
+
'594':
|
|
6
|
+
pendant: pendant light
|
|
7
|
+
sconce: wall sconce
|
|
8
|
+
bollard: Bollard light Bollard Lights British Hand-Made Driveway Lights Path Lights
|
|
9
|
+
Patio Lights Wooden Lights £250-£500
|
|
10
|
+
'604':
|
|
11
|
+
whitegoods: WHITEGOODS Brand:Beko LimitedStock
|
|
12
|
+
'262':
|
|
13
|
+
telco: TELCO SERVICES Brand:Samsung InStock
|
|
14
|
+
'187':
|
|
15
|
+
boots: leather boots
|
|
@@ -21,7 +21,8 @@ module Portage
|
|
|
21
21
|
PRODUCT_PATH = %r{/products?/[^/?#]+}
|
|
22
22
|
|
|
23
23
|
# @param rows [Array<Hash>] Readers rows, all for one domain.
|
|
24
|
-
# @return [Hash{String => Integer}] category id => weight, top 5
|
|
24
|
+
# @return [Hash{String => Integer}] category id => weight, top 5 (equal
|
|
25
|
+
# weights in the order first seen).
|
|
25
26
|
def self.domain(rows)
|
|
26
27
|
texts = Hash.new(0)
|
|
27
28
|
rows.each { |row| texts[text_of(row)] += row[:visits] }
|
|
@@ -29,7 +30,9 @@ module Portage
|
|
|
29
30
|
texts.each do |text, visits|
|
|
30
31
|
Classifier.categories_for(text).each { |id| tally[id] += visits } unless text.empty?
|
|
31
32
|
end
|
|
32
|
-
|
|
33
|
+
# `sort_by` isn't stable, so the position breaks ties: first seen wins.
|
|
34
|
+
tally.each_with_index.sort_by { |(_id, weight), seen| [-weight, seen] }.first(TOP_CATEGORIES)
|
|
35
|
+
.to_h { |pair, _seen| pair }
|
|
33
36
|
end
|
|
34
37
|
|
|
35
38
|
# @param kept [Array<Hash>] Importer's kept entries, rows included.
|
data/lib/portage/cli/check.rb
CHANGED
|
@@ -57,11 +57,21 @@ module Portage
|
|
|
57
57
|
report = report.merge(adapter: adapter_for(report))
|
|
58
58
|
report = report.merge(webmcp: webmcp_for(report))
|
|
59
59
|
verdict = verdict_for(report)
|
|
60
|
-
report.merge(verdict: verdict, next_step: CheckNextStep.call(verdict, report))
|
|
60
|
+
with_index_hint(report.merge(verdict: verdict, next_step: CheckNextStep.call(verdict, report)))
|
|
61
61
|
end
|
|
62
62
|
|
|
63
63
|
private
|
|
64
64
|
|
|
65
|
+
# docs/plans/local-catalogue.md Phase 2: a Shopify (or native UCP)
|
|
66
|
+
# store's catalogue can be crawled into the local index. Check only
|
|
67
|
+
# names the command; it never crawls.
|
|
68
|
+
def with_index_hint(report)
|
|
69
|
+
return report unless report[:native_ucp] || report[:platform] == "Shopify"
|
|
70
|
+
|
|
71
|
+
port = @uri.port == @uri.default_port ? "" : ":#{@uri.port}"
|
|
72
|
+
report.merge(index_hint: "portage index add #{@uri.scheme}://#{@uri.host}#{port} --crawl")
|
|
73
|
+
end
|
|
74
|
+
|
|
65
75
|
def handoff_only_report
|
|
66
76
|
{ url: @uri.to_s, native_ucp: nil, platform: nil, recommended_gem: nil, handoff_only: true, adapter: nil,
|
|
67
77
|
webmcp: webmcp_skipped("#{@uri.host} is hand-off only, so it isn't probed"),
|
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
module Portage
|
|
2
|
+
module Cli
|
|
3
|
+
module Classifier
|
|
4
|
+
# Scores the tokenized input against every node and keeps the best few
|
|
5
|
+
# (docs/plans/local-catalogue.md, Phase 5). Nodes carry hundreds of
|
|
6
|
+
# rolled-up keywords, so this is where "a generic word matches a dozen
|
|
7
|
+
# categories" gets dealt with.
|
|
8
|
+
module Ranking
|
|
9
|
+
# A node's own name words and its descendants' (`keywords`) say what
|
|
10
|
+
# the product is; its parent's (`parent_keywords`, e.g. "home",
|
|
11
|
+
# "garden") only where it is filed, so they count half as much.
|
|
12
|
+
KEYWORD_WEIGHT = 2
|
|
13
|
+
PARENT_WEIGHT = 1
|
|
14
|
+
|
|
15
|
+
# Callers treat every returned id as evidence (store routing, a
|
|
16
|
+
# browser domain's category tally), so the answer keeps only the best
|
|
17
|
+
# MAX_CATEGORIES ids and, of those, only ones scoring at least
|
|
18
|
+
# 1/CUTOFF of the best: "electric kettle" is Kitchen & Dining, not
|
|
19
|
+
# also Chairs because of "electric".
|
|
20
|
+
MAX_CATEGORIES = 3
|
|
21
|
+
CUTOFF = 2
|
|
22
|
+
|
|
23
|
+
module_function
|
|
24
|
+
|
|
25
|
+
# @param counts [Hash{String => Integer}] word => times the input says it.
|
|
26
|
+
# @param table [Classifier::Table]
|
|
27
|
+
# @param stopped [Hash] the stoplist.
|
|
28
|
+
# @return [Array<String>] up to MAX_CATEGORIES node ids, best first.
|
|
29
|
+
def best(counts, table, stopped)
|
|
30
|
+
ranked = score(counts, table).map do |id, (score, evidence)|
|
|
31
|
+
[id, score, evidence, *tie_breakers(table.nodes[id], counts.keys, stopped)]
|
|
32
|
+
end
|
|
33
|
+
ranked = ranked.sort_by { |(_id, score, _evidence, *rest)| [-score.round(6), *rest] }
|
|
34
|
+
cut(ranked).first(MAX_CATEGORIES).map(&:first)
|
|
35
|
+
end
|
|
36
|
+
|
|
37
|
+
# The first row always stays; later rows only if strong?.
|
|
38
|
+
def cut(ranked)
|
|
39
|
+
strongest = ranked.map { |row| row[2] }.max
|
|
40
|
+
ranked.each_with_index.select { |row, i| i.zero? || strong?(row[2], strongest) }.map(&:first)
|
|
41
|
+
end
|
|
42
|
+
|
|
43
|
+
# Distinct input words and how often each occurs. A plural variant of
|
|
44
|
+
# a word counts as the same word ("pendant" and "Pendants").
|
|
45
|
+
def word_counts(tokens)
|
|
46
|
+
tokens.each_with_object({}) do |token, counts|
|
|
47
|
+
word = counts.keys.find { |known| Classifier.word_match?(token, known) } || token
|
|
48
|
+
counts[word] = counts.fetch(word, 0) + 1
|
|
49
|
+
end
|
|
50
|
+
end
|
|
51
|
+
|
|
52
|
+
# @return [Hash{String => Array(Float, Float)}] node id => [score,
|
|
53
|
+
# evidence], for nodes scoring above zero. Each word counts once for
|
|
54
|
+
# a node, however many of its keywords match ("light" and "lights"
|
|
55
|
+
# are two keywords but one word of the input), so a node cannot win
|
|
56
|
+
# by listing a word's variants: KEYWORD_WEIGHT when a `keywords`
|
|
57
|
+
# entry matches, PARENT_WEIGHT when only a `parent_keywords` entry
|
|
58
|
+
# does, times how often the input says the word (1 + ln count: a
|
|
59
|
+
# Shopify tag list repeats "Pendant Lights" once per room, which is
|
|
60
|
+
# its best evidence, and the log keeps a long tag list from burying
|
|
61
|
+
# a different word). `evidence` is the same sum with each word also
|
|
62
|
+
# weighted by how rare it is (ln(1 + nodes / nodes it matches)), so
|
|
63
|
+
# "kettle" counts for more than "electric".
|
|
64
|
+
def score(counts, table)
|
|
65
|
+
scores = Hash.new { |hash, id| hash[id] = [0.0, 0.0] }
|
|
66
|
+
counts.each do |word, count|
|
|
67
|
+
contributions(word, count, table).each { |id, weight, rarity| add(scores[id], weight, rarity) }
|
|
68
|
+
end
|
|
69
|
+
scores
|
|
70
|
+
end
|
|
71
|
+
|
|
72
|
+
# @return [Array<Array(String, Float, Float)>] [node id, weight,
|
|
73
|
+
# rarity] for every node `word` matches.
|
|
74
|
+
def contributions(word, count, table)
|
|
75
|
+
own = matching_ids(word, table.own)
|
|
76
|
+
parent = matching_ids(word, table.parent) - own
|
|
77
|
+
return [] if own.empty? && parent.empty?
|
|
78
|
+
|
|
79
|
+
tf = 1 + Math.log(count)
|
|
80
|
+
rarity = Math.log(1 + table.nodes.size.fdiv((own + parent).length))
|
|
81
|
+
own.map { |id| [id, KEYWORD_WEIGHT * tf, rarity] } + parent.map { |id| [id, PARENT_WEIGHT * tf, rarity] }
|
|
82
|
+
end
|
|
83
|
+
|
|
84
|
+
def add(pair, weight, rarity)
|
|
85
|
+
pair[0] += weight
|
|
86
|
+
pair[1] += weight * rarity
|
|
87
|
+
end
|
|
88
|
+
|
|
89
|
+
# Ranking by `score` alone is what classifies a long tag list best,
|
|
90
|
+
# because rare words are mostly noise there (a room name in a
|
|
91
|
+
# lighting store's tags). The cut uses `evidence`, which a generic
|
|
92
|
+
# word cannot fill: an id beyond the first stays only if its evidence
|
|
93
|
+
# reaches 1/CUTOFF of the strongest.
|
|
94
|
+
def strong?(evidence, strongest) = (evidence * CUTOFF) - strongest > -1e-9
|
|
95
|
+
|
|
96
|
+
# What separates nodes with equal scores, best first: the share of
|
|
97
|
+
# the node's own name the input covers ("Sofas" before "Sofa
|
|
98
|
+
# Accessories" for "sofa"), then a name with no stoplisted word in it
|
|
99
|
+
# ("Household Appliances" before "Household Appliance Accessories"),
|
|
100
|
+
# then fewer keywords (the smaller node is the more specific one),
|
|
101
|
+
# then the file's own order.
|
|
102
|
+
def tie_breakers(node, words, stopped)
|
|
103
|
+
name = node["name"].to_s.split(" > ").last.to_s.downcase.split(/[^\p{Alpha}]+/)
|
|
104
|
+
.select { |word| word.length >= Classifier::MIN_WORD_LENGTH }
|
|
105
|
+
covered = name.count { |part| words.any? { |word| Classifier.word_match?(word, part) } }
|
|
106
|
+
generic = name.any? { |part| stopped.key?(part) } ? 1 : 0
|
|
107
|
+
[-covered.fdiv([name.length, 1].max), generic, Array(node["keywords"]).length, node["order"]]
|
|
108
|
+
end
|
|
109
|
+
|
|
110
|
+
# @return [Array<String>] ids of the nodes in `index` with a keyword
|
|
111
|
+
# `word` matches.
|
|
112
|
+
def matching_ids(word, index)
|
|
113
|
+
keywords_matching(word).flat_map { |keyword| index.fetch(keyword, []) }.uniq
|
|
114
|
+
end
|
|
115
|
+
|
|
116
|
+
# The keywords Classifier.word_match? accepts `word` for: itself, the
|
|
117
|
+
# singular it is a plural of ("boots" -> "boot", "watches" ->
|
|
118
|
+
# "watch"), the plural of it ("boot" -> "boots"), and the "y"/"ies"
|
|
119
|
+
# swap. Looking a word's few candidate keywords up in a hash, rather
|
|
120
|
+
# than comparing every word with every keyword, keeps a long
|
|
121
|
+
# product-tag text (Index::Sources::StorefrontProducts,
|
|
122
|
+
# docs/plans/local-catalogue.md Phase 2) cheap now that nodes carry
|
|
123
|
+
# hundreds of keywords.
|
|
124
|
+
def keywords_matching(word)
|
|
125
|
+
keywords = [word, "#{word}s", "#{word}es"]
|
|
126
|
+
keywords << word.delete_suffix("s") if word.end_with?("s")
|
|
127
|
+
keywords << word.delete_suffix("es") if word.end_with?("es")
|
|
128
|
+
keywords << "#{word[0..-4]}y" if word.end_with?("ies")
|
|
129
|
+
keywords << "#{word[0..-2]}ies" if word.end_with?("y")
|
|
130
|
+
keywords
|
|
131
|
+
end
|
|
132
|
+
end
|
|
133
|
+
end
|
|
134
|
+
end
|
|
135
|
+
end
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
require "yaml"
|
|
2
|
+
|
|
3
|
+
module Portage
|
|
4
|
+
module Cli
|
|
5
|
+
module Classifier
|
|
6
|
+
# The shipped categories.yml and the user's ~/.portage/categories.yml,
|
|
7
|
+
# merged (a user node replaces a shipped node of the same id), with the
|
|
8
|
+
# keyword => node ids indexes the scoring looks words up in.
|
|
9
|
+
Table = Struct.new(:nodes, :own, :parent)
|
|
10
|
+
|
|
11
|
+
class Table
|
|
12
|
+
@cache = {}
|
|
13
|
+
|
|
14
|
+
class << self
|
|
15
|
+
# Built once per pair of files and kept for the process, keyed by
|
|
16
|
+
# the files' mtime and size, so an edit to ~/.portage/categories.yml
|
|
17
|
+
# still shows up on the next call; classifying a whole catalogue
|
|
18
|
+
# would otherwise rebuild the index for every product.
|
|
19
|
+
def for(known_path, user_path)
|
|
20
|
+
key = [signature(known_path), signature(user_path)]
|
|
21
|
+
@cache.clear if @cache.size > 8
|
|
22
|
+
@cache[key] ||= build(known_path, user_path)
|
|
23
|
+
end
|
|
24
|
+
|
|
25
|
+
# @return [Hash] the YAML mapping at `path`; empty when the file is
|
|
26
|
+
# missing, unreadable or not a mapping.
|
|
27
|
+
def load_yaml(path)
|
|
28
|
+
return {} unless path && File.readable?(path)
|
|
29
|
+
|
|
30
|
+
data = YAML.safe_load_file(path)
|
|
31
|
+
data.is_a?(Hash) ? data : {}
|
|
32
|
+
rescue StandardError
|
|
33
|
+
{}
|
|
34
|
+
end
|
|
35
|
+
|
|
36
|
+
private
|
|
37
|
+
|
|
38
|
+
def build(known_path, user_path)
|
|
39
|
+
ordered = {}
|
|
40
|
+
load_yaml(known_path).each_with_index { |(id, node), i| ordered[id] = node.merge("order" => i) }
|
|
41
|
+
load_yaml(user_path).each_with_index do |(id, node), i|
|
|
42
|
+
ordered[id] = node.merge("order" => ordered.size + i)
|
|
43
|
+
end
|
|
44
|
+
new(ordered, index(ordered, "keywords"), index(ordered, "parent_keywords"))
|
|
45
|
+
end
|
|
46
|
+
|
|
47
|
+
def index(nodes, field)
|
|
48
|
+
result = Hash.new { |hash, key| hash[key] = [] }
|
|
49
|
+
nodes.each { |id, node| Array(node[field]).each { |keyword| result[keyword] << id } }
|
|
50
|
+
result.default_proc = nil
|
|
51
|
+
result
|
|
52
|
+
end
|
|
53
|
+
|
|
54
|
+
def signature(path)
|
|
55
|
+
return nil unless path && File.readable?(path)
|
|
56
|
+
|
|
57
|
+
[path, File.mtime(path).to_r, File.size(path)]
|
|
58
|
+
end
|
|
59
|
+
end
|
|
60
|
+
end
|
|
61
|
+
end
|
|
62
|
+
end
|
|
63
|
+
end
|
|
@@ -1,4 +1,6 @@
|
|
|
1
1
|
require "yaml"
|
|
2
|
+
require_relative "classifier/table"
|
|
3
|
+
require_relative "classifier/ranking"
|
|
2
4
|
|
|
3
5
|
module Portage
|
|
4
6
|
module Cli
|
|
@@ -18,10 +20,21 @@ module Portage
|
|
|
18
20
|
# check matches in both directions ("carpet" contains "pet", "chair"
|
|
19
21
|
# contains "hair", "scarf" contains "car") and was a real source of
|
|
20
22
|
# false positives before this became whole-word.
|
|
23
|
+
#
|
|
24
|
+
# `known-stores/categories.yml` is generated by `script/categories`: a level-2
|
|
25
|
+
# node's `keywords` are its own name plus its descendants', `parent_keywords`
|
|
26
|
+
# are its parent's words, and `category-stoplist.yml` words are left out.
|
|
27
|
+
# See Ranking for how a text is scored against it.
|
|
21
28
|
module Classifier
|
|
22
29
|
# `known-stores/categories.yml`, from `lib/portage/cli/classifier.rb`.
|
|
23
30
|
KNOWN_PATH = File.expand_path("../../../known-stores/categories.yml", __dir__).freeze
|
|
24
31
|
|
|
32
|
+
# Words that are never evidence of a category (`known-stores/category-stoplist.yml`,
|
|
33
|
+
# `word: why`): merchandising and url words, and words that name a whole family of
|
|
34
|
+
# nodes. `script/categories` leaves them out of the keywords; `categories_for`
|
|
35
|
+
# drops them from its input, so the two sides agree.
|
|
36
|
+
STOPLIST_PATH = File.expand_path("../../../known-stores/category-stoplist.yml", __dir__).freeze
|
|
37
|
+
|
|
25
38
|
# The user's own additions/overrides — same id overrides a shipped
|
|
26
39
|
# node's keywords, a new id extends the taxonomy. Absent by default;
|
|
27
40
|
# nothing here is required for the shipped file to work.
|
|
@@ -44,21 +57,22 @@ module Portage
|
|
|
44
57
|
# @param known_path [String] override for KNOWN_PATH — specs redirect
|
|
45
58
|
# this the same way SearchBackends::Allowlist takes its own `path:`.
|
|
46
59
|
# @param user_path [String] override for PATH.
|
|
47
|
-
# @
|
|
48
|
-
#
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
60
|
+
# @param stoplist_path [String] override for STOPLIST_PATH.
|
|
61
|
+
# @return [Array<String>] up to MAX_CATEGORIES category ids, best score
|
|
62
|
+
# first (ties: see .tie_breakers). Empty when nothing matches.
|
|
63
|
+
def self.categories_for(text, known_path: KNOWN_PATH, user_path: PATH, stoplist_path: STOPLIST_PATH)
|
|
64
|
+
stopped = Table.load_yaml(stoplist_path)
|
|
65
|
+
counts = Ranking.word_counts(tokenize(text).reject { |word| stopped.key?(word) })
|
|
66
|
+
return [] if counts.empty?
|
|
67
|
+
|
|
68
|
+
Ranking.best(counts, Table.for(known_path, user_path), stopped)
|
|
55
69
|
end
|
|
56
70
|
|
|
57
71
|
# @return [Array<String>] the taxonomy names for `ids`, in order —
|
|
58
72
|
# `portage browser import` (Phase 3) shows a domain's guessed
|
|
59
73
|
# categories by name so the user can judge them before saving.
|
|
60
74
|
def self.names_for(ids, known_path: KNOWN_PATH, user_path: PATH)
|
|
61
|
-
all =
|
|
75
|
+
all = Table.for(known_path, user_path).nodes
|
|
62
76
|
Array(ids).filter_map { |id| all.dig(id.to_s, "name") }
|
|
63
77
|
end
|
|
64
78
|
|
|
@@ -88,17 +102,6 @@ module Portage
|
|
|
88
102
|
.select { |word| word.length >= MIN_WORD_LENGTH }
|
|
89
103
|
end
|
|
90
104
|
|
|
91
|
-
# --- Scoring one node against the tokenized input ---
|
|
92
|
-
|
|
93
|
-
def self.rank(id, node, words)
|
|
94
|
-
keywords = Array(node["keywords"])
|
|
95
|
-
hits = keywords.count { |keyword| words.any? { |word| word_match?(word, keyword) } }
|
|
96
|
-
return nil unless hits.positive?
|
|
97
|
-
|
|
98
|
-
[id, hits, node["order"].to_i]
|
|
99
|
-
end
|
|
100
|
-
private_class_method :rank
|
|
101
|
-
|
|
102
105
|
# Whole-word only, plus the plural forms a keyword list and a real
|
|
103
106
|
# query/title actually differ by: an exact match, one plus a trailing
|
|
104
107
|
# "s" or "es" ("boot"/"boots", "watch"/"watches"), or the "y"/"ies"
|
|
@@ -129,30 +132,6 @@ module Portage
|
|
|
129
132
|
(keyword.end_with?("ies") && word == "#{keyword[0..-4]}y")
|
|
130
133
|
end
|
|
131
134
|
private_class_method :ies_y_match?
|
|
132
|
-
|
|
133
|
-
# --- Loading and merging the two files ---
|
|
134
|
-
|
|
135
|
-
# Re-read on every call rather than cached process-wide: `find` calls
|
|
136
|
-
# this once or twice per invocation, not in a hot loop, and a cached
|
|
137
|
-
# copy would miss an edit to ~/.portage/categories.yml until the next
|
|
138
|
-
# process start.
|
|
139
|
-
def self.nodes(known_path, user_path)
|
|
140
|
-
ordered = {}
|
|
141
|
-
load_yaml(known_path).each_with_index { |(id, node), i| ordered[id] = node.merge("order" => i) }
|
|
142
|
-
load_yaml(user_path).each_with_index { |(id, node), i| ordered[id] = node.merge("order" => ordered.size + i) }
|
|
143
|
-
ordered
|
|
144
|
-
end
|
|
145
|
-
private_class_method :nodes
|
|
146
|
-
|
|
147
|
-
def self.load_yaml(path)
|
|
148
|
-
return {} unless path && File.readable?(path)
|
|
149
|
-
|
|
150
|
-
data = YAML.safe_load_file(path)
|
|
151
|
-
data.is_a?(Hash) ? data : {}
|
|
152
|
-
rescue StandardError
|
|
153
|
-
{}
|
|
154
|
-
end
|
|
155
|
-
private_class_method :load_yaml
|
|
156
135
|
end
|
|
157
136
|
end
|
|
158
137
|
end
|
data/lib/portage/cli/doctor.rb
CHANGED
|
@@ -42,7 +42,8 @@ module Portage
|
|
|
42
42
|
# those four warnings were noise on every fresh install.
|
|
43
43
|
def initialize(adapter_class: nil, proxy_settings: ProxySettings.new, install_doctor: InstallDoctor.new,
|
|
44
44
|
seller: true, dot_env_path: DotEnv.loaded_path, index_stores: Index::Store.new,
|
|
45
|
-
index_products: Index::ProductStore.new, known_cache: Index::KnownCache.new
|
|
45
|
+
index_products: Index::ProductStore.new, known_cache: Index::KnownCache.new,
|
|
46
|
+
index_database: Index::Database.new(path: Index::Database.path_for(Index::Store::PATH)))
|
|
46
47
|
@adapter_class = adapter_class
|
|
47
48
|
@proxy_settings = proxy_settings
|
|
48
49
|
@install_doctor = install_doctor
|
|
@@ -51,6 +52,7 @@ module Portage
|
|
|
51
52
|
@index_stores = index_stores
|
|
52
53
|
@index_products = index_products
|
|
53
54
|
@known_cache = known_cache
|
|
55
|
+
@index_database = index_database
|
|
54
56
|
end
|
|
55
57
|
|
|
56
58
|
# `portage setup` always offers its wizard on a TTY; a bare `portage
|
|
@@ -210,12 +212,20 @@ module Portage
|
|
|
210
212
|
# leaves behind.
|
|
211
213
|
def index_finding
|
|
212
214
|
refresh_known_cache_if_stale
|
|
213
|
-
|
|
214
|
-
|
|
215
|
+
first = @index_stores.exists? ? index_message : no_index_message
|
|
216
|
+
database = @index_database.info
|
|
217
|
+
Finding.new(check: "index", level: "info", details: { database: database },
|
|
218
|
+
message: [first, database_message(database), known_cache_message].join("\n"))
|
|
219
|
+
end
|
|
220
|
+
|
|
221
|
+
def no_index_message
|
|
222
|
+
"No local index yet — run `portage index build` to give `find` a list of " \
|
|
223
|
+
"stores/products on top of stores.yml and web search."
|
|
224
|
+
end
|
|
215
225
|
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
226
|
+
def database_message(info)
|
|
227
|
+
"Index database: #{info[:path]} (#{info[:stores]} store row(s), #{info[:products]} product row(s), " \
|
|
228
|
+
"FTS5 #{info[:fts5] ? 'available' : 'not available'})."
|
|
219
229
|
end
|
|
220
230
|
|
|
221
231
|
def refresh_known_cache_if_stale
|
data/lib/portage/cli/find.rb
CHANGED
|
@@ -229,9 +229,11 @@ module Portage
|
|
|
229
229
|
amount, currency = price_of(product)
|
|
230
230
|
return nil if over_max_price?(amount)
|
|
231
231
|
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
|
|
232
|
+
OfferSources.with_product(
|
|
233
|
+
{ store: store[:origin], source: store[:source], checkout: store[:checkout],
|
|
234
|
+
product_id: field(product, "id"), title: field(product, "title"),
|
|
235
|
+
amount: amount, currency: currency, url: field(product, "url") }, product
|
|
236
|
+
)
|
|
235
237
|
end
|
|
236
238
|
|
|
237
239
|
# An unpriced offer stays in: no price isn't the same as too dear.
|
|
@@ -37,6 +37,9 @@ module Portage
|
|
|
37
37
|
THROTTLE = 0.1
|
|
38
38
|
STALE_AFTER = 7 * 24 * 60 * 60
|
|
39
39
|
TOP_CATEGORIES = 5
|
|
40
|
+
# Product sightings per ProductStore#upsert_many transaction — one
|
|
41
|
+
# products.json page's worth (docs/plans/local-catalogue.md Phase 2).
|
|
42
|
+
WRITE_BATCH = 250
|
|
40
43
|
|
|
41
44
|
# Public so BrowserImport::Importer (Phase 3) labels a probed
|
|
42
45
|
# origin's capabilities exactly the way an index build does.
|
|
@@ -94,13 +97,19 @@ module Portage
|
|
|
94
97
|
# HandoffOnly) is recorded without ever probing it — the user
|
|
95
98
|
# explicitly named it, but that's still not a request this process
|
|
96
99
|
# sends.
|
|
97
|
-
|
|
100
|
+
#
|
|
101
|
+
# `crawl: true` (`index add URL --crawl`) then reads the store's own
|
|
102
|
+
# catalogue through Sources::StorefrontProducts into the index.
|
|
103
|
+
# Opt-in: a crawl is up to 21 more requests and 20s of pauses, where
|
|
104
|
+
# a plain add is one probe.
|
|
105
|
+
def add(url, crawl: false)
|
|
98
106
|
origin = origin_of(url)
|
|
99
107
|
return { added: false, message: "Not a valid http(s) URL: #{url}" } unless origin
|
|
100
108
|
return store_manual_handoff_only(origin) if handoff_only_origin?(origin)
|
|
101
109
|
|
|
102
110
|
session = probe(origin)
|
|
103
|
-
store_manual(origin, session)
|
|
111
|
+
result = store_manual(origin, session)
|
|
112
|
+
crawl ? crawl_added(origin, result) : result
|
|
104
113
|
end
|
|
105
114
|
|
|
106
115
|
# `portage index remove HOST`
|
|
@@ -184,10 +193,29 @@ module Portage
|
|
|
184
193
|
{ added: true, origin: origin, message: "Added #{origin} — hand-off only, never probed." }
|
|
185
194
|
end
|
|
186
195
|
|
|
196
|
+
# `store_fields:` on a sighting (StorefrontProducts' crawl note and
|
|
197
|
+
# platform) lands on the store row as-is.
|
|
187
198
|
def update_existing(origin, group)
|
|
188
199
|
existing = @stores.find(origin)
|
|
200
|
+
fields = group.filter_map { |g| g[:store_fields] }.reduce({}, :merge)
|
|
189
201
|
@stores.upsert(origin, sources: merged_sources(existing, group),
|
|
190
|
-
categories: merge_categories(existing["categories"], group))
|
|
202
|
+
categories: merge_categories(existing["categories"], group), **fields)
|
|
203
|
+
end
|
|
204
|
+
|
|
205
|
+
def crawl_added(origin, result)
|
|
206
|
+
sightings = Sources::StorefrontProducts.new(stores: @stores, handoff_only: @handoff_only)
|
|
207
|
+
.crawl(origin, platform: @stores.find(origin)&.dig("platform"))
|
|
208
|
+
tagged = sightings.map { |s| s.merge(source: "storefront_products") }
|
|
209
|
+
apply(tagged, dry_run: false)
|
|
210
|
+
note = tagged.last.dig(:store_fields, :crawl)
|
|
211
|
+
result.merge(crawl: note, message: "#{result[:message]} #{crawl_message(note)}")
|
|
212
|
+
end
|
|
213
|
+
|
|
214
|
+
def crawl_message(note)
|
|
215
|
+
return "Catalogue not crawled (#{note['reason']})." if note["status"] == "skipped"
|
|
216
|
+
|
|
217
|
+
"Crawled #{note['products']} product(s) from #{note['pages']} page(s)" \
|
|
218
|
+
"#{" (stopped: #{note['reason']})" if note['reason']}."
|
|
191
219
|
end
|
|
192
220
|
|
|
193
221
|
def store_new(origin, session, group)
|
|
@@ -220,7 +248,7 @@ module Portage
|
|
|
220
248
|
group.each do |sighting|
|
|
221
249
|
next unless sighting[:title]
|
|
222
250
|
|
|
223
|
-
|
|
251
|
+
categories_of(sighting).each { |id| tally[id] += 1 }
|
|
224
252
|
end
|
|
225
253
|
tally.sort_by { |_id, weight| -weight }.first(TOP_CATEGORIES).to_h
|
|
226
254
|
end
|
|
@@ -232,17 +260,26 @@ module Portage
|
|
|
232
260
|
def capabilities_of(session) = self.class.capabilities_of(session)
|
|
233
261
|
|
|
234
262
|
def store_products(sightings)
|
|
235
|
-
|
|
236
|
-
eligible.
|
|
263
|
+
known = Hash.new { |memo, origin| memo[origin] = !@stores.find(origin).nil? }
|
|
264
|
+
eligible = sightings.select { |s| s[:title] && known[s[:origin]] }
|
|
265
|
+
eligible.each_slice(WRITE_BATCH) { |batch| @products.upsert_many(batch.map { |s| product_row(s) }) }
|
|
237
266
|
eligible.length
|
|
238
267
|
end
|
|
239
268
|
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
269
|
+
# `product:` on a sighting (StorefrontProducts' handle, url,
|
|
270
|
+
# image_url, options, variant_ids) is stored alongside the usual
|
|
271
|
+
# fields.
|
|
272
|
+
def product_row(sighting)
|
|
273
|
+
{ key: product_key(sighting), origin: sighting[:origin], seen_at: @now.to_i, title: sighting[:title],
|
|
274
|
+
brand: sighting[:brand], gtin: sighting[:gtin], category: categories_of(sighting).first,
|
|
275
|
+
sources: [sighting[:source]].compact, **sighting.fetch(:product, {}) }
|
|
276
|
+
end
|
|
277
|
+
|
|
278
|
+
# A source that already classified its sighting (StorefrontProducts,
|
|
279
|
+
# on product_type and tags) says so in `categories:`; otherwise the
|
|
280
|
+
# title is classified here.
|
|
281
|
+
def categories_of(sighting)
|
|
282
|
+
sighting[:categories] || Classifier.categories_for(sighting[:title])
|
|
246
283
|
end
|
|
247
284
|
|
|
248
285
|
# GTIN when a source has one (none do yet); otherwise a normalized
|