portage-cli 0.10.0 → 0.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +129 -0
- data/README.md +107 -21
- data/known-stores/categories.yml +6841 -18
- data/known-stores/category-stoplist.yml +40 -0
- data/known-stores/category-synonyms.yml +15 -0
- data/lib/portage/cli/browser_import/categorize.rb +5 -2
- data/lib/portage/cli/buy.rb +217 -39
- data/lib/portage/cli/check.rb +11 -1
- data/lib/portage/cli/classifier/ranking.rb +135 -0
- data/lib/portage/cli/classifier/table.rb +63 -0
- data/lib/portage/cli/classifier.rb +23 -44
- data/lib/portage/cli/confidence_check.rb +27 -7
- data/lib/portage/cli/confidence_state.rb +109 -0
- data/lib/portage/cli/doctor.rb +16 -6
- data/lib/portage/cli/find.rb +69 -4
- data/lib/portage/cli/index/builder.rb +49 -12
- data/lib/portage/cli/index/database.rb +150 -0
- data/lib/portage/cli/index/entry_product.rb +37 -0
- data/lib/portage/cli/index/legacy_import.rb +54 -0
- data/lib/portage/cli/index/product_store.rb +73 -35
- data/lib/portage/cli/index/schema.rb +70 -0
- data/lib/portage/cli/index/search.rb +73 -0
- data/lib/portage/cli/index/sources/storefront_products/mapper.rb +127 -0
- data/lib/portage/cli/index/sources/storefront_products/pages.rb +114 -0
- data/lib/portage/cli/index/sources/storefront_products/robots.rb +70 -0
- data/lib/portage/cli/index/sources/storefront_products.rb +147 -0
- data/lib/portage/cli/index/sources.rb +5 -2
- data/lib/portage/cli/index/store.rb +23 -47
- data/lib/portage/cli/index.rb +1 -0
- data/lib/portage/cli/offer_sources.rb +17 -2
- data/lib/portage/cli/version.rb +1 -1
- data/lib/portage/cli.rb +111 -12
- metadata +30 -2
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
# Words Classifier never uses as evidence of a category, on either side: script/categories
|
|
2
|
+
# leaves them out of categories.yml, and Classifier.categories_for drops them from its
|
|
3
|
+
# input. Each entry is `word: why`. Both forms of a word are listed, because the stoplist is
|
|
4
|
+
# matched exactly (the keyword match is what normalizes plurals).
|
|
5
|
+
#
|
|
6
|
+
# Two kinds of word belong here: words that describe how something is sold or filed rather
|
|
7
|
+
# than what it is, and words that label a whole family of nodes ("accessories" names 76 of
|
|
8
|
+
# the 192 level-2 nodes). A specific word that is merely ambiguous does not belong here.
|
|
9
|
+
and: function word; appears in taxonomy names ("Hardware & Tools")
|
|
10
|
+
for: function word; appears in taxonomy names ("Filters for ...")
|
|
11
|
+
the: function word
|
|
12
|
+
with: function word ("Vac with Multi-Purpose Head")
|
|
13
|
+
from: function word
|
|
14
|
+
new: merchandising flag; Light Yard tags products "New Collection"
|
|
15
|
+
sale: merchandising flag; a sale tag says nothing about the product
|
|
16
|
+
gift: merchandising flag; "Gift Set" is a way to sell, not a product type
|
|
17
|
+
gifts: merchandising flag; plural of gift
|
|
18
|
+
set: bundle word; "Gift Set" and "Furniture Set" say how it is packaged
|
|
19
|
+
sets: bundle word; plural of set
|
|
20
|
+
kit: bundle word; "drill kit" is the drill
|
|
21
|
+
kits: bundle word; plural of kit
|
|
22
|
+
collection: storefront and url word; matched "Toll Collection Devices" (4488) for every "New Collection" tag
|
|
23
|
+
collections: url and storefront word; the /collections/ path segment of every Shopify collection url
|
|
24
|
+
products: url word; the /products/ path segment of every Shopify product url
|
|
25
|
+
product: url and page-title word
|
|
26
|
+
category: url word; the /category/ path segment
|
|
27
|
+
shop: page-title word ("Cordless Drills | Hardware | Shop")
|
|
28
|
+
accessories: names 76 of the 192 level-2 nodes, so it cannot tell them apart
|
|
29
|
+
accessory: singular of accessories
|
|
30
|
+
supplies: names 25 level-2 nodes ("Pet Supplies", "Office Supplies"), so it cannot tell them apart
|
|
31
|
+
supply: singular of supplies
|
|
32
|
+
equipment: names 21 level-2 nodes, so it cannot tell them apart
|
|
33
|
+
parts: names 21 level-2 nodes ("Vehicle Parts"), so it cannot tell them apart
|
|
34
|
+
replacement: a spare, not a product type; "Replacement Fade Blade Set"
|
|
35
|
+
general: filler word ("General Office Supplies")
|
|
36
|
+
other: filler word; the catch-all name in many taxonomies
|
|
37
|
+
miscellaneous: catch-all product type; JB Hi-Fi files 237 of its first 1,250 products under "MISCELLANEOUS"
|
|
38
|
+
service: generic product-type word; JB Hi-Fi's "TELCO SERVICES" type matched Food Service (135)
|
|
39
|
+
services: plural of service; JB Hi-Fi files 279 of its first 1,250 products under "TELCO SERVICES"
|
|
40
|
+
hand: craft marker; Light Yard tags 146 of its 164 products "British Hand-Made", and "hand" also names 10 level-2 nodes ("Hand Tools")
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
# Words the Google taxonomy has no node for, added to a node's keywords by script/categories.
|
|
2
|
+
# `id:` then `word: why`; the why names the golden case (spec/fixtures/classifier_golden.yml)
|
|
3
|
+
# that needs it, by its text. Add a word here only for a case the golden set proves, never
|
|
4
|
+
# in bulk.
|
|
5
|
+
'594':
|
|
6
|
+
pendant: pendant light
|
|
7
|
+
sconce: wall sconce
|
|
8
|
+
bollard: Bollard light Bollard Lights British Hand-Made Driveway Lights Path Lights
|
|
9
|
+
Patio Lights Wooden Lights £250-£500
|
|
10
|
+
'604':
|
|
11
|
+
whitegoods: WHITEGOODS Brand:Beko LimitedStock
|
|
12
|
+
'262':
|
|
13
|
+
telco: TELCO SERVICES Brand:Samsung InStock
|
|
14
|
+
'187':
|
|
15
|
+
boots: leather boots
|
|
@@ -21,7 +21,8 @@ module Portage
|
|
|
21
21
|
PRODUCT_PATH = %r{/products?/[^/?#]+}
|
|
22
22
|
|
|
23
23
|
# @param rows [Array<Hash>] Readers rows, all for one domain.
|
|
24
|
-
# @return [Hash{String => Integer}] category id => weight, top 5
|
|
24
|
+
# @return [Hash{String => Integer}] category id => weight, top 5 (equal
|
|
25
|
+
# weights in the order first seen).
|
|
25
26
|
def self.domain(rows)
|
|
26
27
|
texts = Hash.new(0)
|
|
27
28
|
rows.each { |row| texts[text_of(row)] += row[:visits] }
|
|
@@ -29,7 +30,9 @@ module Portage
|
|
|
29
30
|
texts.each do |text, visits|
|
|
30
31
|
Classifier.categories_for(text).each { |id| tally[id] += visits } unless text.empty?
|
|
31
32
|
end
|
|
32
|
-
|
|
33
|
+
# `sort_by` isn't stable, so the position breaks ties: first seen wins.
|
|
34
|
+
tally.each_with_index.sort_by { |(_id, weight), seen| [-weight, seen] }.first(TOP_CATEGORIES)
|
|
35
|
+
.to_h { |pair, _seen| pair }
|
|
33
36
|
end
|
|
34
37
|
|
|
35
38
|
# @param kept [Array<Hash>] Importer's kept entries, rows included.
|
data/lib/portage/cli/buy.rb
CHANGED
|
@@ -9,6 +9,7 @@ require_relative "payment_methods"
|
|
|
9
9
|
require_relative "setting"
|
|
10
10
|
require_relative "decisions"
|
|
11
11
|
require_relative "confidence_check"
|
|
12
|
+
require_relative "confidence_state"
|
|
12
13
|
require_relative "checkout_handoff"
|
|
13
14
|
require_relative "money"
|
|
14
15
|
require_relative "notifier"
|
|
@@ -122,14 +123,18 @@ module Portage
|
|
|
122
123
|
# hand off the real checkout first checks its total against this
|
|
123
124
|
# (with `quote_currency:`) and reports `quote_changed` instead if it
|
|
124
125
|
# is higher, in another currency, or missing. See #finish_checkout.
|
|
125
|
-
#
|
|
126
|
+
# @param quote_store [String, nil] the approved quote's store, and
|
|
127
|
+
# @param quote_title [String, nil] its title — set by `buy --quote`
|
|
128
|
+
# alongside quote_total:. Only the confidence check reads them: they
|
|
129
|
+
# go into its `approved_quote` (see #confidence_quote).
|
|
130
|
+
# rubocop:disable Metrics/ParameterLists, Metrics/MethodLength, Metrics/AbcSize -- all keywords; one per flag, plus
|
|
126
131
|
# injectable collaborators, each assigned to its own ivar
|
|
127
132
|
def initialize(url:, query:, qty: 1, payment_token: nil, yes: false, dry_run: false, product_id: nil,
|
|
128
133
|
auto_open: nil, notify_webhook: nil, handoff_target: nil, confidence_check: nil,
|
|
129
134
|
transaction_log: nil, max_price: nil, webmcp_bridge: nil, webmcp_mappings: nil,
|
|
130
135
|
webmcp_mapping_confirm: nil, autofill: nil, webmcp_autofill_confirm: nil, json: false,
|
|
131
|
-
quote_total: nil, quote_currency: nil)
|
|
132
|
-
# rubocop:enable Metrics/ParameterLists, Metrics/MethodLength
|
|
136
|
+
quote_total: nil, quote_currency: nil, quote_store: nil, quote_title: nil)
|
|
137
|
+
# rubocop:enable Metrics/ParameterLists, Metrics/MethodLength, Metrics/AbcSize
|
|
133
138
|
raw = url.to_s.strip
|
|
134
139
|
@uri = URI.parse(raw =~ %r{\Ahttps?://}i ? raw : "https://#{raw}")
|
|
135
140
|
@query = query
|
|
@@ -151,6 +156,8 @@ module Portage
|
|
|
151
156
|
@json = json
|
|
152
157
|
@quote_total = quote_total
|
|
153
158
|
@quote_currency = quote_currency
|
|
159
|
+
@quote_store = quote_store
|
|
160
|
+
@quote_title = quote_title
|
|
154
161
|
@webmcp_bridge = webmcp_bridge
|
|
155
162
|
@decisions = {}
|
|
156
163
|
end
|
|
@@ -510,7 +517,9 @@ module Portage
|
|
|
510
517
|
# call it against. Builds a cart, reads it back through the (already
|
|
511
518
|
# cart/checkout-capable, per the gate in #webmcp_flow) session so
|
|
512
519
|
# #reconcile_checkout has a `get_cart`-shaped document to check against
|
|
513
|
-
# (there's no checkout document either),
|
|
520
|
+
# (there's no checkout document either), stops there if the quote cap,
|
|
521
|
+
# the mismatch check or the opt-in confidence check holds it
|
|
522
|
+
# (#webmcp_cart_held_report), and only then calls the preset's
|
|
514
523
|
# `handoff_checkout` tool directly on the bridge — after the cart
|
|
515
524
|
# read-back, not before, so any post-mutation "page not ready" gap
|
|
516
525
|
# (see Transport's own retry) has already been waited out by then.
|
|
@@ -536,24 +545,84 @@ module Portage
|
|
|
536
545
|
end
|
|
537
546
|
return webmcp_handoff_dry_run_report(products, product, preset) if @dry_run
|
|
538
547
|
|
|
539
|
-
|
|
540
|
-
|
|
541
|
-
|
|
542
|
-
|
|
548
|
+
@product = product
|
|
549
|
+
cart = webmcp_build_cart(session, product)
|
|
550
|
+
held = webmcp_cart_held_report(products, product, cart)
|
|
551
|
+
return held if held
|
|
543
552
|
|
|
544
553
|
result = @webmcp_bridge.execute_tool(preset.handoff_checkout, {})
|
|
545
554
|
autofill = attempt_webmcp_autofill(preset)
|
|
546
|
-
webmcp_handoff_report("webmcp", products, cart.merge("continue_url" => url_from_handoff(result)),
|
|
555
|
+
webmcp_handoff_report("webmcp", products, cart.merge("continue_url" => url_from_handoff(result)), [],
|
|
547
556
|
autofill: autofill)
|
|
548
557
|
end
|
|
549
558
|
|
|
559
|
+
# The same gates #finish_checkout runs before anything leaves this
|
|
560
|
+
# process, in the same order, all before the hand-off tool sends the
|
|
561
|
+
# tab to checkout and before anything is autofilled: the quote cap,
|
|
562
|
+
# then the deterministic mismatch stop, then (only when a decision
|
|
563
|
+
# backend is enabled) the confidence check. The confidence check only
|
|
564
|
+
# sees a cart the first two let through, so it can hold this hand-off
|
|
565
|
+
# but never let through one they stop.
|
|
566
|
+
# @return [Hash, nil] the report for the first gate that held, or nil.
|
|
567
|
+
def webmcp_cart_held_report(products, product, cart)
|
|
568
|
+
warnings = reconcile_checkout(product, cart)
|
|
569
|
+
return quote_changed_report("webmcp", products, cart, warnings) if quote_exceeded?(cart)
|
|
570
|
+
return webmcp_cart_mismatch_report(products, cart, warnings) if warnings.any?
|
|
571
|
+
|
|
572
|
+
webmcp_low_confidence_report(products, cart)
|
|
573
|
+
end
|
|
574
|
+
|
|
575
|
+
# A hold here points at the store's cart page, like the mismatch stop:
|
|
576
|
+
# checkout was never opened, so there's nothing at a checkout URL yet.
|
|
577
|
+
def webmcp_low_confidence_report(products, cart)
|
|
578
|
+
verdict = decide_confidence(cart, [])
|
|
579
|
+
return nil if verdict.nil? || verdict[:proceed]
|
|
580
|
+
|
|
581
|
+
handoff_report("webmcp", products, cart.merge("continue_url" => webmcp_cart_page_url), [],
|
|
582
|
+
outcome: "low_confidence",
|
|
583
|
+
message: "#{low_confidence_message(verdict)} Checkout wasn't opened, and nothing was " \
|
|
584
|
+
"autofilled.")
|
|
585
|
+
end
|
|
586
|
+
|
|
587
|
+
# Adds the line to the store's cart, then reads the cart back.
|
|
588
|
+
def webmcp_build_cart(session, product)
|
|
589
|
+
created = session.create_cart(line_items: [{ product_id: line_item_id_of(product), quantity: @qty }],
|
|
590
|
+
context: buyer_context, meta: agent_meta)
|
|
591
|
+
session.get_cart(cart_id: created["id"], meta: agent_meta)
|
|
592
|
+
end
|
|
593
|
+
|
|
594
|
+
# Same fail-closed rule as #full_buy (see #decide_escalation): a cart
|
|
595
|
+
# that doesn't match the request stops here, before the hand-off tool
|
|
596
|
+
# sends the tab to checkout and before anything is autofilled. This
|
|
597
|
+
# flow used to only add the mismatch to `warnings` and carry on to
|
|
598
|
+
# checkout. The cart already exists on the store, so the report points
|
|
599
|
+
# at the store's cart page for the person to inspect, not at checkout.
|
|
600
|
+
# Every preset with a `handoff_checkout` tool today is Shopify's, where
|
|
601
|
+
# that page is /cart. A cart has no checkout status, so the verdict is
|
|
602
|
+
# the mismatch alone: `decisions.escalation.reason` is "mismatch", as
|
|
603
|
+
# on #full_buy's own stop.
|
|
604
|
+
def webmcp_cart_mismatch_report(products, cart, warnings)
|
|
605
|
+
@decisions[:escalation] = Decisions.escalation(checkout_status: nil, warnings: warnings)
|
|
606
|
+
handoff_report("webmcp", products, cart.merge("continue_url" => webmcp_cart_page_url), warnings,
|
|
607
|
+
outcome: "checkout_mismatch",
|
|
608
|
+
message: "Stopped before checkout — the store's cart didn't match the request: " \
|
|
609
|
+
"#{warnings.join(' ')} Nothing was bought, and checkout wasn't opened.")
|
|
610
|
+
end
|
|
611
|
+
|
|
612
|
+
def webmcp_cart_page_url
|
|
613
|
+
URI.join(@uri, "/cart").to_s
|
|
614
|
+
end
|
|
615
|
+
|
|
550
616
|
# Unlike #full_buy's dry run (which still creates a UCP checkout, since
|
|
551
617
|
# that's a document the store drops on its own), a dry run here stops
|
|
552
618
|
# before `create_cart`: every step after it changes something outside
|
|
553
619
|
# this process — a real cart on the store, the bridge's own tab
|
|
554
620
|
# navigated to checkout, and (with autofill on) typing into that page.
|
|
555
621
|
# None of that is safe to do on a preview run, so this only reports
|
|
556
|
-
# what the run would have done.
|
|
622
|
+
# what the run would have done. That also means there's no cart to
|
|
623
|
+
# reconcile, so unlike #dry_run_report this can never carry
|
|
624
|
+
# `checkout_mismatch: true`; the real run checks the cart and stops
|
|
625
|
+
# there instead. No checkout_id either, so History
|
|
557
626
|
# records it as a search, not a purchase.
|
|
558
627
|
def webmcp_handoff_dry_run_report(products, product, preset)
|
|
559
628
|
build_report(
|
|
@@ -764,6 +833,7 @@ module Portage
|
|
|
764
833
|
message: no_match_message)
|
|
765
834
|
end
|
|
766
835
|
|
|
836
|
+
@product = product
|
|
767
837
|
checkout = session.create_checkout(line_items: [{ product_id: line_item_id_of(product), quantity: @qty }],
|
|
768
838
|
fulfillment: requested_fulfillment(fulfillment_adapter),
|
|
769
839
|
context: buyer_context, meta: agent_meta)
|
|
@@ -779,34 +849,78 @@ module Portage
|
|
|
779
849
|
# (if surprising) checkout response, not a transport error. Buying
|
|
780
850
|
# blind against a mismatch this method could have caught defeats the
|
|
781
851
|
# point of an agent shopping on the buyer's behalf, so this always
|
|
782
|
-
# checks and
|
|
783
|
-
#
|
|
852
|
+
# checks, and any warning it returns stops a real purchase before
|
|
853
|
+
# payment (see #decide_escalation). The quoted total and currency are
|
|
854
|
+
# #quote_exceeded?'s job, not this method's.
|
|
855
|
+
#
|
|
856
|
+
# This is the authoritative check, and it always runs first: the
|
|
857
|
+
# opt-in model check (#decide_confidence) only ever sees a checkout
|
|
858
|
+
# that already passed it, and can hold that checkout but never
|
|
859
|
+
# overrule a warning from here.
|
|
784
860
|
def reconcile_checkout(product, checkout)
|
|
785
861
|
item_id = line_item_id_of(product)
|
|
786
|
-
|
|
862
|
+
lines = Array(checkout["line_items"])
|
|
863
|
+
line = lines.find { |li| li.dig("item", "id") == item_id }
|
|
787
864
|
return ["Store dropped the requested item (#{item_id}) from checkout."] unless line
|
|
788
865
|
|
|
789
866
|
warnings = []
|
|
790
867
|
if line["quantity"] != @qty
|
|
791
868
|
warnings << "Store checked out quantity #{line['quantity']}, not the requested #{@qty}."
|
|
792
869
|
end
|
|
870
|
+
warnings + price_mismatches(product, item_id, line, checkout["currency"]) +
|
|
871
|
+
unrequested_lines(lines.reject { |li| li.equal?(line) })
|
|
872
|
+
end
|
|
873
|
+
|
|
874
|
+
# Only one line is ever requested, so any other line is something the
|
|
875
|
+
# person never asked for: an upsell, a "shipping protection" add-on,
|
|
876
|
+
# or (on a WebMCP cart) whatever was already sitting in the store's
|
|
877
|
+
# cart. One that costs nothing is let through, so a store's free gift
|
|
878
|
+
# or $0 sample doesn't stop the purchase; one whose cost can't be read
|
|
879
|
+
# is treated as costing something, so it stops.
|
|
880
|
+
def unrequested_lines(extras)
|
|
881
|
+
extras.reject { |li| line_cost(li)&.zero? }.map do |li|
|
|
882
|
+
label = [li.dig("item", "id"), li.dig("item", "title")].compact.join(" ")
|
|
883
|
+
"Store added #{label.empty? ? 'a line' : label} to checkout, which wasn't requested."
|
|
884
|
+
end
|
|
885
|
+
end
|
|
886
|
+
|
|
887
|
+
# The line's own total, else its unit price times its quantity; nil
|
|
888
|
+
# when the line carries neither.
|
|
889
|
+
def line_cost(line)
|
|
890
|
+
total = Portage::Ucp::Support::Totals.amount(line["totals"])
|
|
891
|
+
return total if total.is_a?(Numeric)
|
|
892
|
+
|
|
893
|
+
price = line.dig("item", "price")
|
|
894
|
+
quantity = line["quantity"] || 1
|
|
895
|
+
price * quantity if price.is_a?(Numeric) && quantity.is_a?(Numeric)
|
|
896
|
+
end
|
|
897
|
+
|
|
898
|
+
# A unit price is only comparable in the catalog's own currency, so a
|
|
899
|
+
# checkout in another one is reported as that, not as a price change.
|
|
900
|
+
def price_mismatches(product, item_id, line, currency)
|
|
901
|
+
catalog_currency = expected_unit_currency(product, item_id)
|
|
902
|
+
if catalog_currency && currency && catalog_currency != currency
|
|
903
|
+
return ["Store checked out in #{currency}, not the catalog's #{catalog_currency}."]
|
|
904
|
+
end
|
|
793
905
|
|
|
794
906
|
expected = expected_unit_price(product, item_id)
|
|
795
907
|
actual = line.dig("item", "price")
|
|
796
|
-
|
|
797
|
-
|
|
798
|
-
|
|
799
|
-
end
|
|
800
|
-
warnings
|
|
908
|
+
return [] unless expected && actual && expected != actual
|
|
909
|
+
|
|
910
|
+
["Store priced the item at #{actual} #{currency} minor units per unit, not the catalog's #{expected}."]
|
|
801
911
|
end
|
|
802
912
|
|
|
803
913
|
def expected_unit_price(product, item_id)
|
|
804
|
-
|
|
805
|
-
variant&.dig("price", "amount") || product.dig("price_range", "min", "amount")
|
|
914
|
+
expected_price_of(product, item_id)&.dig("amount")
|
|
806
915
|
end
|
|
807
916
|
|
|
808
|
-
def
|
|
809
|
-
|
|
917
|
+
def expected_unit_currency(product, item_id)
|
|
918
|
+
expected_price_of(product, item_id)&.dig("currency")
|
|
919
|
+
end
|
|
920
|
+
|
|
921
|
+
def expected_price_of(product, item_id)
|
|
922
|
+
variant = Array(product["variants"]).find { |v| v["id"] == item_id }
|
|
923
|
+
variant&.dig("price") || product.dig("price_range", "min")
|
|
810
924
|
end
|
|
811
925
|
|
|
812
926
|
# Submits PORTAGE_SHIP_* (see Portage::Cli::ShippingProfile) as the
|
|
@@ -963,22 +1077,38 @@ module Portage
|
|
|
963
1077
|
escalation = decide_escalation(checkout, warnings)
|
|
964
1078
|
return escalated_report(source, products, checkout, warnings, escalation) if escalation[:escalate]
|
|
965
1079
|
return dry_run_report(source, products, checkout, warnings) if @dry_run
|
|
966
|
-
return
|
|
1080
|
+
return webmcp_handoff_checkout_report(source, products, checkout, warnings) if force_handoff
|
|
967
1081
|
return confirmation_needed_report(source, products, checkout, warnings) unless confirmed?
|
|
968
1082
|
|
|
969
1083
|
complete(session, source, products, checkout, warnings)
|
|
970
1084
|
end
|
|
971
1085
|
|
|
1086
|
+
# The non-preset WebMCP hand-off (`express_stop` through #full_buy). Same
|
|
1087
|
+
# last gate as #webmcp_handoff_checkout_flow's: with a decision backend
|
|
1088
|
+
# enabled, the confidence check runs before anything is handed off, on a
|
|
1089
|
+
# checkout the quote cap and the mismatch stop already let through. A
|
|
1090
|
+
# hold reports `low_confidence` pointing at the store's cart page, as the
|
|
1091
|
+
# preset flow's does, so nothing is sent to a `profile` target's checkout.
|
|
1092
|
+
# No backend named: straight to the hand-off, as before.
|
|
1093
|
+
def webmcp_handoff_checkout_report(source, products, checkout, warnings)
|
|
1094
|
+
webmcp_low_confidence_report(products, checkout) ||
|
|
1095
|
+
webmcp_handoff_report(source, products, checkout, warnings)
|
|
1096
|
+
end
|
|
1097
|
+
|
|
972
1098
|
# First in #finish_checkout, ahead of the escalation gates: a
|
|
973
1099
|
# `quote_changed` refusal never hands off, since that would spend the
|
|
974
1100
|
# quote and open a checkout the person never approved. #complete is
|
|
975
1101
|
# the only place this file charges, and it is reached only through
|
|
976
1102
|
# #finish_checkout.
|
|
1103
|
+
#
|
|
1104
|
+
# A quote with no total (its dry run had none to show, as a WebMCP
|
|
1105
|
+
# preset dry run never does) caps nothing, so it refuses too, rather
|
|
1106
|
+
# than buying at whatever the store now asks.
|
|
977
1107
|
def quote_exceeded?(checkout)
|
|
978
|
-
return false unless
|
|
1108
|
+
return false unless quote_run?
|
|
979
1109
|
|
|
980
1110
|
total = checkout_total(checkout)
|
|
981
|
-
total.nil? || total > @quote_total || checkout["currency"] != @quote_currency
|
|
1111
|
+
@quote_total.nil? || total.nil? || total > @quote_total || checkout["currency"] != @quote_currency
|
|
982
1112
|
end
|
|
983
1113
|
|
|
984
1114
|
def quote_changed_report(source, products, checkout, warnings)
|
|
@@ -990,6 +1120,11 @@ module Portage
|
|
|
990
1120
|
end
|
|
991
1121
|
|
|
992
1122
|
def quote_changed_message(total, currency)
|
|
1123
|
+
if @quote_total.nil?
|
|
1124
|
+
return "The quote has no total to hold this checkout (now #{quoted_amount(total, currency)}) to — " \
|
|
1125
|
+
"nothing was bought. Dry-run again for a priced quote."
|
|
1126
|
+
end
|
|
1127
|
+
|
|
993
1128
|
"The price changed since the quote (was #{quoted_amount(@quote_total, @quote_currency)}, " \
|
|
994
1129
|
"now #{quoted_amount(total, currency)}) — nothing was bought."
|
|
995
1130
|
end
|
|
@@ -998,15 +1133,18 @@ module Portage
|
|
|
998
1133
|
|
|
999
1134
|
# Hand off vs. keep going is Decisions.escalation's call
|
|
1000
1135
|
# (docs/plans/system-one-decision-layer.md § Responsibilities 2): a
|
|
1001
|
-
# literal `requires_escalation` status always escalates
|
|
1002
|
-
# from #reconcile_checkout
|
|
1003
|
-
#
|
|
1004
|
-
#
|
|
1136
|
+
# literal `requires_escalation` status always escalates, and so does
|
|
1137
|
+
# any mismatch from #reconcile_checkout, with no setting to turn that
|
|
1138
|
+
# off. This used to only warn unless PORTAGE_ABORT_ON_CHECKOUT_MISMATCH
|
|
1139
|
+
# was set, so by default a checkout that didn't match what the person
|
|
1140
|
+
# approved was still paid for and reported `purchased`. That variable
|
|
1141
|
+
# is now ignored. A --dry-run never pays, so there the mismatch stays
|
|
1142
|
+
# in `warnings` and #dry_run_report flags it instead, since an agent
|
|
1143
|
+
# needs the priced checkout to tell the person what went wrong.
|
|
1005
1144
|
# The verdict lands on the report's `decisions:`, and the report's
|
|
1006
1145
|
# `outcome:` names which gate (if any) stopped the purchase.
|
|
1007
1146
|
def decide_escalation(checkout, warnings)
|
|
1008
|
-
verdict = Decisions.escalation(checkout_status: checkout["status"],
|
|
1009
|
-
warnings: abort_on_mismatch? ? warnings : [])
|
|
1147
|
+
verdict = Decisions.escalation(checkout_status: checkout["status"], warnings: @dry_run ? [] : warnings)
|
|
1010
1148
|
@decisions[:escalation] = verdict
|
|
1011
1149
|
verdict
|
|
1012
1150
|
end
|
|
@@ -1129,17 +1267,48 @@ module Portage
|
|
|
1129
1267
|
Portage::Ucp::Support::Totals.amount(checkout["totals"])
|
|
1130
1268
|
end
|
|
1131
1269
|
|
|
1132
|
-
#
|
|
1133
|
-
#
|
|
1270
|
+
# Runs only on a checkout that already passed #reconcile_checkout, the
|
|
1271
|
+
# quote cap and the spend policy, so it can hold a purchase but never
|
|
1272
|
+
# let through one those would stop. The state is ConfidenceState's
|
|
1273
|
+
# allowlisted summary — never the payment token, the address or the
|
|
1274
|
+
# buyer's contact details — because it goes to a model backend that
|
|
1275
|
+
# may be a hosted third-party API (Jev).
|
|
1134
1276
|
def decide_confidence(checkout, warnings)
|
|
1135
|
-
|
|
1136
|
-
|
|
1137
|
-
|
|
1138
|
-
)
|
|
1277
|
+
return nil unless confidence_check.enabled?
|
|
1278
|
+
|
|
1279
|
+
verdict = confidence_check.call(confidence_state(checkout, warnings))
|
|
1139
1280
|
@decisions[:confidence] = verdict if verdict
|
|
1140
1281
|
verdict
|
|
1141
1282
|
end
|
|
1142
1283
|
|
|
1284
|
+
def confidence_state(checkout, warnings)
|
|
1285
|
+
item_id = @product && line_item_id_of(@product)
|
|
1286
|
+
ConfidenceState.build(
|
|
1287
|
+
request: { query: @query, merchant: @uri.host, quantity: @qty, item_id: item_id,
|
|
1288
|
+
item_title: picked_title(item_id) },
|
|
1289
|
+
checkout: checkout, warnings: warnings, quote: confidence_quote
|
|
1290
|
+
)
|
|
1291
|
+
end
|
|
1292
|
+
|
|
1293
|
+
# The picked product's title from the store's own search, plus its
|
|
1294
|
+
# variant's when the checked-out variant has one ("Tee — Large").
|
|
1295
|
+
def picked_title(item_id)
|
|
1296
|
+
return nil unless @product
|
|
1297
|
+
|
|
1298
|
+
[@product["title"], variant_matching(@product, item_id)&.dig("title")].compact.uniq.join(" — ")
|
|
1299
|
+
end
|
|
1300
|
+
|
|
1301
|
+
# What `buy --quote` pinned, as the person approved it. nil on a run
|
|
1302
|
+
# with no quote.
|
|
1303
|
+
def confidence_quote
|
|
1304
|
+
return nil unless quote_run?
|
|
1305
|
+
|
|
1306
|
+
{ store: @quote_store, product_id: @product_id, title: @quote_title, quantity: @qty,
|
|
1307
|
+
total: @quote_total, currency: @quote_currency }
|
|
1308
|
+
end
|
|
1309
|
+
|
|
1310
|
+
def quote_run? = !(@quote_store || @quote_total || @quote_currency).nil?
|
|
1311
|
+
|
|
1143
1312
|
def confidence_check
|
|
1144
1313
|
@confidence_check ||= ConfidenceCheck.new
|
|
1145
1314
|
end
|
|
@@ -1410,9 +1579,18 @@ module Portage
|
|
|
1410
1579
|
@yes
|
|
1411
1580
|
end
|
|
1412
1581
|
|
|
1582
|
+
# `warnings` here are #reconcile_checkout's mismatches, which a real
|
|
1583
|
+
# run of the same checkout stops on — so the report says so up front,
|
|
1584
|
+
# before anyone approves a quote for it.
|
|
1413
1585
|
def dry_run_report(source, products, checkout, warnings = [])
|
|
1414
|
-
|
|
1415
|
-
|
|
1586
|
+
message = "Dry run — checkout created but not completed."
|
|
1587
|
+
extra = {}
|
|
1588
|
+
if warnings.any?
|
|
1589
|
+
message += " It doesn't match the request (#{warnings.join(' ')}), so a real purchase would stop " \
|
|
1590
|
+
"with checkout_mismatch."
|
|
1591
|
+
extra[:checkout_mismatch] = true
|
|
1592
|
+
end
|
|
1593
|
+
checkout_report(source, products, checkout, outcome: "dry_run", warnings: warnings, message: message, **extra)
|
|
1416
1594
|
end
|
|
1417
1595
|
|
|
1418
1596
|
def confirmation_needed_report(source, products, checkout, warnings = [])
|
data/lib/portage/cli/check.rb
CHANGED
|
@@ -57,11 +57,21 @@ module Portage
|
|
|
57
57
|
report = report.merge(adapter: adapter_for(report))
|
|
58
58
|
report = report.merge(webmcp: webmcp_for(report))
|
|
59
59
|
verdict = verdict_for(report)
|
|
60
|
-
report.merge(verdict: verdict, next_step: CheckNextStep.call(verdict, report))
|
|
60
|
+
with_index_hint(report.merge(verdict: verdict, next_step: CheckNextStep.call(verdict, report)))
|
|
61
61
|
end
|
|
62
62
|
|
|
63
63
|
private
|
|
64
64
|
|
|
65
|
+
# docs/plans/local-catalogue.md Phase 2: a Shopify (or native UCP)
|
|
66
|
+
# store's catalogue can be crawled into the local index. Check only
|
|
67
|
+
# names the command; it never crawls.
|
|
68
|
+
def with_index_hint(report)
|
|
69
|
+
return report unless report[:native_ucp] || report[:platform] == "Shopify"
|
|
70
|
+
|
|
71
|
+
port = @uri.port == @uri.default_port ? "" : ":#{@uri.port}"
|
|
72
|
+
report.merge(index_hint: "portage index add #{@uri.scheme}://#{@uri.host}#{port} --crawl")
|
|
73
|
+
end
|
|
74
|
+
|
|
65
75
|
def handoff_only_report
|
|
66
76
|
{ url: @uri.to_s, native_ucp: nil, platform: nil, recommended_gem: nil, handoff_only: true, adapter: nil,
|
|
67
77
|
webmcp: webmcp_skipped("#{@uri.host} is hand-off only, so it isn't probed"),
|
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
module Portage
|
|
2
|
+
module Cli
|
|
3
|
+
module Classifier
|
|
4
|
+
# Scores the tokenized input against every node and keeps the best few
|
|
5
|
+
# (docs/plans/local-catalogue.md, Phase 5). Nodes carry hundreds of
|
|
6
|
+
# rolled-up keywords, so this is where "a generic word matches a dozen
|
|
7
|
+
# categories" gets dealt with.
|
|
8
|
+
module Ranking
|
|
9
|
+
# A node's own name words and its descendants' (`keywords`) say what
|
|
10
|
+
# the product is; its parent's (`parent_keywords`, e.g. "home",
|
|
11
|
+
# "garden") only where it is filed, so they count half as much.
|
|
12
|
+
KEYWORD_WEIGHT = 2
|
|
13
|
+
PARENT_WEIGHT = 1
|
|
14
|
+
|
|
15
|
+
# Callers treat every returned id as evidence (store routing, a
|
|
16
|
+
# browser domain's category tally), so the answer keeps only the best
|
|
17
|
+
# MAX_CATEGORIES ids and, of those, only ones scoring at least
|
|
18
|
+
# 1/CUTOFF of the best: "electric kettle" is Kitchen & Dining, not
|
|
19
|
+
# also Chairs because of "electric".
|
|
20
|
+
MAX_CATEGORIES = 3
|
|
21
|
+
CUTOFF = 2
|
|
22
|
+
|
|
23
|
+
module_function
|
|
24
|
+
|
|
25
|
+
# @param counts [Hash{String => Integer}] word => times the input says it.
|
|
26
|
+
# @param table [Classifier::Table]
|
|
27
|
+
# @param stopped [Hash] the stoplist.
|
|
28
|
+
# @return [Array<String>] up to MAX_CATEGORIES node ids, best first.
|
|
29
|
+
def best(counts, table, stopped)
|
|
30
|
+
ranked = score(counts, table).map do |id, (score, evidence)|
|
|
31
|
+
[id, score, evidence, *tie_breakers(table.nodes[id], counts.keys, stopped)]
|
|
32
|
+
end
|
|
33
|
+
ranked = ranked.sort_by { |(_id, score, _evidence, *rest)| [-score.round(6), *rest] }
|
|
34
|
+
cut(ranked).first(MAX_CATEGORIES).map(&:first)
|
|
35
|
+
end
|
|
36
|
+
|
|
37
|
+
# The first row always stays; later rows only if strong?.
|
|
38
|
+
def cut(ranked)
|
|
39
|
+
strongest = ranked.map { |row| row[2] }.max
|
|
40
|
+
ranked.each_with_index.select { |row, i| i.zero? || strong?(row[2], strongest) }.map(&:first)
|
|
41
|
+
end
|
|
42
|
+
|
|
43
|
+
# Distinct input words and how often each occurs. A plural variant of
|
|
44
|
+
# a word counts as the same word ("pendant" and "Pendants").
|
|
45
|
+
def word_counts(tokens)
|
|
46
|
+
tokens.each_with_object({}) do |token, counts|
|
|
47
|
+
word = counts.keys.find { |known| Classifier.word_match?(token, known) } || token
|
|
48
|
+
counts[word] = counts.fetch(word, 0) + 1
|
|
49
|
+
end
|
|
50
|
+
end
|
|
51
|
+
|
|
52
|
+
# @return [Hash{String => Array(Float, Float)}] node id => [score,
|
|
53
|
+
# evidence], for nodes scoring above zero. Each word counts once for
|
|
54
|
+
# a node, however many of its keywords match ("light" and "lights"
|
|
55
|
+
# are two keywords but one word of the input), so a node cannot win
|
|
56
|
+
# by listing a word's variants: KEYWORD_WEIGHT when a `keywords`
|
|
57
|
+
# entry matches, PARENT_WEIGHT when only a `parent_keywords` entry
|
|
58
|
+
# does, times how often the input says the word (1 + ln count: a
|
|
59
|
+
# Shopify tag list repeats "Pendant Lights" once per room, which is
|
|
60
|
+
# its best evidence, and the log keeps a long tag list from burying
|
|
61
|
+
# a different word). `evidence` is the same sum with each word also
|
|
62
|
+
# weighted by how rare it is (ln(1 + nodes / nodes it matches)), so
|
|
63
|
+
# "kettle" counts for more than "electric".
|
|
64
|
+
def score(counts, table)
|
|
65
|
+
scores = Hash.new { |hash, id| hash[id] = [0.0, 0.0] }
|
|
66
|
+
counts.each do |word, count|
|
|
67
|
+
contributions(word, count, table).each { |id, weight, rarity| add(scores[id], weight, rarity) }
|
|
68
|
+
end
|
|
69
|
+
scores
|
|
70
|
+
end
|
|
71
|
+
|
|
72
|
+
# @return [Array<Array(String, Float, Float)>] [node id, weight,
|
|
73
|
+
# rarity] for every node `word` matches.
|
|
74
|
+
def contributions(word, count, table)
|
|
75
|
+
own = matching_ids(word, table.own)
|
|
76
|
+
parent = matching_ids(word, table.parent) - own
|
|
77
|
+
return [] if own.empty? && parent.empty?
|
|
78
|
+
|
|
79
|
+
tf = 1 + Math.log(count)
|
|
80
|
+
rarity = Math.log(1 + table.nodes.size.fdiv((own + parent).length))
|
|
81
|
+
own.map { |id| [id, KEYWORD_WEIGHT * tf, rarity] } + parent.map { |id| [id, PARENT_WEIGHT * tf, rarity] }
|
|
82
|
+
end
|
|
83
|
+
|
|
84
|
+
def add(pair, weight, rarity)
|
|
85
|
+
pair[0] += weight
|
|
86
|
+
pair[1] += weight * rarity
|
|
87
|
+
end
|
|
88
|
+
|
|
89
|
+
# Ranking by `score` alone is what classifies a long tag list best,
|
|
90
|
+
# because rare words are mostly noise there (a room name in a
|
|
91
|
+
# lighting store's tags). The cut uses `evidence`, which a generic
|
|
92
|
+
# word cannot fill: an id beyond the first stays only if its evidence
|
|
93
|
+
# reaches 1/CUTOFF of the strongest.
|
|
94
|
+
def strong?(evidence, strongest) = (evidence * CUTOFF) - strongest > -1e-9
|
|
95
|
+
|
|
96
|
+
# What separates nodes with equal scores, best first: the share of
|
|
97
|
+
# the node's own name the input covers ("Sofas" before "Sofa
|
|
98
|
+
# Accessories" for "sofa"), then a name with no stoplisted word in it
|
|
99
|
+
# ("Household Appliances" before "Household Appliance Accessories"),
|
|
100
|
+
# then fewer keywords (the smaller node is the more specific one),
|
|
101
|
+
# then the file's own order.
|
|
102
|
+
def tie_breakers(node, words, stopped)
|
|
103
|
+
name = node["name"].to_s.split(" > ").last.to_s.downcase.split(/[^\p{Alpha}]+/)
|
|
104
|
+
.select { |word| word.length >= Classifier::MIN_WORD_LENGTH }
|
|
105
|
+
covered = name.count { |part| words.any? { |word| Classifier.word_match?(word, part) } }
|
|
106
|
+
generic = name.any? { |part| stopped.key?(part) } ? 1 : 0
|
|
107
|
+
[-covered.fdiv([name.length, 1].max), generic, Array(node["keywords"]).length, node["order"]]
|
|
108
|
+
end
|
|
109
|
+
|
|
110
|
+
# @return [Array<String>] ids of the nodes in `index` with a keyword
|
|
111
|
+
# `word` matches.
|
|
112
|
+
def matching_ids(word, index)
|
|
113
|
+
keywords_matching(word).flat_map { |keyword| index.fetch(keyword, []) }.uniq
|
|
114
|
+
end
|
|
115
|
+
|
|
116
|
+
# The keywords Classifier.word_match? accepts `word` for: itself, the
|
|
117
|
+
# singular it is a plural of ("boots" -> "boot", "watches" ->
|
|
118
|
+
# "watch"), the plural of it ("boot" -> "boots"), and the "y"/"ies"
|
|
119
|
+
# swap. Looking a word's few candidate keywords up in a hash, rather
|
|
120
|
+
# than comparing every word with every keyword, keeps a long
|
|
121
|
+
# product-tag text (Index::Sources::StorefrontProducts,
|
|
122
|
+
# docs/plans/local-catalogue.md Phase 2) cheap now that nodes carry
|
|
123
|
+
# hundreds of keywords.
|
|
124
|
+
def keywords_matching(word)
|
|
125
|
+
keywords = [word, "#{word}s", "#{word}es"]
|
|
126
|
+
keywords << word.delete_suffix("s") if word.end_with?("s")
|
|
127
|
+
keywords << word.delete_suffix("es") if word.end_with?("es")
|
|
128
|
+
keywords << "#{word[0..-4]}y" if word.end_with?("ies")
|
|
129
|
+
keywords << "#{word[0..-2]}ies" if word.end_with?("y")
|
|
130
|
+
keywords
|
|
131
|
+
end
|
|
132
|
+
end
|
|
133
|
+
end
|
|
134
|
+
end
|
|
135
|
+
end
|