portage-cli 0.7.5 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +559 -0
  3. data/README.md +266 -5
  4. data/known-stores/categories.yml +1263 -0
  5. data/lib/portage/cli/agent_profile_url.rb +30 -0
  6. data/lib/portage/cli/browser_import/categorize.rb +59 -0
  7. data/lib/portage/cli/browser_import/confirm.rb +35 -0
  8. data/lib/portage/cli/browser_import/domains.rb +47 -0
  9. data/lib/portage/cli/browser_import/filter.rb +91 -0
  10. data/lib/portage/cli/browser_import/importer.rb +248 -0
  11. data/lib/portage/cli/browser_import/plist_xml.rb +72 -0
  12. data/lib/portage/cli/browser_import/prober.rb +60 -0
  13. data/lib/portage/cli/browser_import/profiles.rb +114 -0
  14. data/lib/portage/cli/browser_import/readers.rb +179 -0
  15. data/lib/portage/cli/browser_import/saver.rb +62 -0
  16. data/lib/portage/cli/browser_import/sqlite.rb +68 -0
  17. data/lib/portage/cli/browser_import.rb +23 -0
  18. data/lib/portage/cli/browser_profile/allowlist.rb +40 -0
  19. data/lib/portage/cli/browser_profile/bridge.rb +120 -0
  20. data/lib/portage/cli/browser_profile/browsers.rb +69 -0
  21. data/lib/portage/cli/browser_profile/cdp.rb +67 -0
  22. data/lib/portage/cli/browser_profile/cdp_socket.rb +186 -0
  23. data/lib/portage/cli/browser_profile/errors.rb +26 -0
  24. data/lib/portage/cli/browser_profile/launcher.rb +34 -0
  25. data/lib/portage/cli/browser_profile/profile.rb +93 -0
  26. data/lib/portage/cli/browser_profile.rb +25 -0
  27. data/lib/portage/cli/buy.rb +521 -29
  28. data/lib/portage/cli/classifier.rb +158 -0
  29. data/lib/portage/cli/compare.rb +3 -0
  30. data/lib/portage/cli/doctor.rb +155 -1
  31. data/lib/portage/cli/dot_env.rb +55 -0
  32. data/lib/portage/cli/find.rb +96 -12
  33. data/lib/portage/cli/handoff_agents.rb +186 -0
  34. data/lib/portage/cli/handoff_only.rb +94 -0
  35. data/lib/portage/cli/handoff_reconciler.rb +15 -1
  36. data/lib/portage/cli/handoff_target.rb +61 -0
  37. data/lib/portage/cli/index/builder.rb +335 -0
  38. data/lib/portage/cli/index/exporter.rb +91 -0
  39. data/lib/portage/cli/index/known_cache.rb +155 -0
  40. data/lib/portage/cli/index/product_store.rb +101 -0
  41. data/lib/portage/cli/index/sources/browser.rb +31 -0
  42. data/lib/portage/cli/index/sources/shopify_catalog.rb +82 -0
  43. data/lib/portage/cli/index/sources/stores_file.rb +58 -0
  44. data/lib/portage/cli/index/sources/webmcp_sweep.rb +29 -0
  45. data/lib/portage/cli/index/sources/wikidata.rb +95 -0
  46. data/lib/portage/cli/index/sources.rb +44 -0
  47. data/lib/portage/cli/index/store.rb +109 -0
  48. data/lib/portage/cli/index.rb +20 -0
  49. data/lib/portage/cli/known_stores_url.rb +15 -0
  50. data/lib/portage/cli/offer_sources.rb +460 -0
  51. data/lib/portage/cli/payment_methods.rb +24 -3
  52. data/lib/portage/cli/search_backends.rb +337 -12
  53. data/lib/portage/cli/setup_wizard/prompt.rb +67 -0
  54. data/lib/portage/cli/setup_wizard/steps/agent_profile.rb +60 -0
  55. data/lib/portage/cli/setup_wizard/steps/browser_import.rb +29 -0
  56. data/lib/portage/cli/setup_wizard/steps/handoff.rb +100 -0
  57. data/lib/portage/cli/setup_wizard/steps/index_build.rb +31 -0
  58. data/lib/portage/cli/setup_wizard/steps/policy.rb +60 -0
  59. data/lib/portage/cli/setup_wizard/steps/retailer_keys.rb +55 -0
  60. data/lib/portage/cli/setup_wizard/steps/search_keys.rb +54 -0
  61. data/lib/portage/cli/setup_wizard/steps/shipping.rb +51 -0
  62. data/lib/portage/cli/setup_wizard.rb +74 -0
  63. data/lib/portage/cli/version.rb +1 -1
  64. data/lib/portage/cli/webmcp.rb +10 -3
  65. data/lib/portage/cli/webmcp_autofill_confirm.rb +38 -0
  66. data/lib/portage/cli/webmcp_autofill_fields.rb +60 -0
  67. data/lib/portage/cli/webmcp_autofill_mode.rb +39 -0
  68. data/lib/portage/cli/webmcp_mapping_confirm.rb +68 -0
  69. data/lib/portage/cli/webmcp_mappings.rb +84 -0
  70. data/lib/portage/cli.rb +525 -5
  71. metadata +58 -2
@@ -0,0 +1,30 @@
1
+ module Portage
2
+ module Cli
3
+ # Where `find`/`buy`/`OfferSources::ShopifyCatalog` get
4
+ # `meta.ucp-agent.profile` when the caller hasn't set
5
+ # PORTAGE_AGENT_PROFILE.
6
+ #
7
+ # Real UCP servers fetch and verify that URL before answering any call
8
+ # (Transports::Http), and the CLI used to read it via a bare
9
+ # `ENV.fetch("PORTAGE_AGENT_PROFILE", nil)` with no fallback — a fresh
10
+ # install that never copied `.env.example` got a bare
11
+ # MissingAgentProfileError on its very first `find`/`buy`. The repo
12
+ # already publishes its own profile document
13
+ # (`portage-cli/agent-profile/agent-profile.json`) over jsdelivr, and
14
+ # that URL is confirmed live against catalog.shopify.com (see
15
+ # docs/agent-profile.md), so it's a reasonable "works enough to try it"
16
+ # default — not a substitute for a caller hosting their own via
17
+ # `portage generate agent-profile`.
18
+ module AgentProfileUrl
19
+ DEFAULT = "https://cdn.jsdelivr.net/gh/tomtom87/Portage@main/portage-cli/agent-profile/agent-profile.json"
20
+ .freeze
21
+
22
+ # @return [String] PORTAGE_AGENT_PROFILE, or DEFAULT when it's unset
23
+ # or blank.
24
+ def self.resolve
25
+ value = ENV.fetch("PORTAGE_AGENT_PROFILE", nil).to_s.strip
26
+ value.empty? ? DEFAULT : value
27
+ end
28
+ end
29
+ end
30
+ end
@@ -0,0 +1,59 @@
1
+ require_relative "../classifier"
2
+ require_relative "../index/builder"
3
+
4
+ module Portage
5
+ module Cli
6
+ module BrowserImport
7
+ # "Smarter import" (docs/plans/buy-skill-and-local-browser.md Phase
8
+ # 3): a kept domain's categories come from what the user actually
9
+ # looked at there — each row's page title, bookmark folder name and
10
+ # URL slug (`/products/<slug>`, `/collections/<slug>`, `/c/<slug>`,
11
+ # `/category/<slug>`, split on `-`/`_`), run through the same
12
+ # Classifier every other index source uses, and weighted by visit
13
+ # count. A domain none of whose rows matches a category stays
14
+ # uncategorised: SearchBackends::Index then routes it by name only,
15
+ # never for a generic query.
16
+ module Categorize
17
+ TOP_CATEGORIES = Index::Builder::TOP_CATEGORIES
18
+
19
+ # A product page, for `--include-product-pages`: `/product/<x>` or
20
+ # `/products/<x>` (Shopify, WooCommerce and most others).
21
+ PRODUCT_PATH = %r{/products?/[^/?#]+}
22
+
23
+ # @param rows [Array<Hash>] Readers rows, all for one domain.
24
+ # @return [Hash{String => Integer}] category id => weight, top 5.
25
+ def self.domain(rows)
26
+ texts = Hash.new(0)
27
+ rows.each { |row| texts[text_of(row)] += row[:visits] }
28
+ tally = Hash.new(0)
29
+ texts.each do |text, visits|
30
+ Classifier.categories_for(text).each { |id| tally[id] += visits } unless text.empty?
31
+ end
32
+ tally.sort_by { |_id, weight| -weight }.first(TOP_CATEGORIES).to_h
33
+ end
34
+
35
+ # @param kept [Array<Hash>] Importer's kept entries, rows included.
36
+ # @return [Array<Hash>] one product entry per distinct product-page
37
+ # title — key, title, url, origin, source ("history"/"bookmark"),
38
+ # category.
39
+ def self.products(kept)
40
+ kept.flat_map { |entry| products_of(entry) }.uniq { |p| p[:key] }
41
+ end
42
+
43
+ def self.products_of(entry)
44
+ entry[:rows].select { |r| r[:url].match?(PRODUCT_PATH) && !r[:title].empty? }.map do |row|
45
+ { key: "title:#{row[:title].downcase.gsub(/[^a-z0-9]+/, '-')}", title: row[:title], url: row[:url],
46
+ origin: entry[:origin], source: row[:kind], category: Classifier.categories_for(text_of(row)).first }
47
+ end
48
+ end
49
+ private_class_method :products_of
50
+
51
+ def self.text_of(row)
52
+ slugs = Classifier::SLUG_PATTERNS.filter_map { |pattern| pattern.match(row[:url])&.[](1) }
53
+ [row[:title], row[:folder], *slugs.map { |s| s.tr("-_", " ") }].reject(&:empty?).join(" ")
54
+ end
55
+ private_class_method :text_of
56
+ end
57
+ end
58
+ end
59
+ end
@@ -0,0 +1,35 @@
1
+ module Portage
2
+ module Cli
3
+ module BrowserImport
4
+ # The "shows the list and asks before writing" gate (docs/plans/
5
+ # buy-skill-and-local-browser.md Phase 3), shaped like
6
+ # WebmcpMappingConfirm: an explicit `--yes` is the only way to save
7
+ # without a prompt; a prompt needs a real TTY and no `--json`; and
8
+ # anything else — piped, CI, an agent running `--json` — saves
9
+ # nothing and says how to confirm. A dry run never saves, `--yes` or
10
+ # not.
11
+ class Confirm
12
+ # @param interactive [Boolean] false under --json or with no TTY on
13
+ # stdin (the caller decides).
14
+ def initialize(interactive:, input: $stdin, output: $stdout)
15
+ @interactive = interactive
16
+ @input = input
17
+ @output = output
18
+ end
19
+
20
+ # @return [Symbol] :save, :declined, :dry_run, :nothing, or
21
+ # :needs_confirmation (no TTY/--json and no --yes).
22
+ def call(plan, yes:, dry_run:)
23
+ return :dry_run if dry_run
24
+ return :nothing if Array(plan[:kept]).empty?
25
+ return :save if yes
26
+ return :needs_confirmation unless @interactive
27
+
28
+ @output.print "Save these #{plan[:kept].length} store(s) to your local index? [y/N] "
29
+ @output.flush
30
+ @input.gets.to_s.strip.downcase == "y" ? :save : :declined
31
+ end
32
+ end
33
+ end
34
+ end
35
+ end
@@ -0,0 +1,47 @@
1
+ require "uri"
2
+
3
+ module Portage
4
+ module Cli
5
+ module BrowserImport
6
+ # Reduces Readers rows to domains — the only thing about a row an
7
+ # import keeps past this point, besides the page title/folder/slug
8
+ # words Categorize reads locally. `www.` is folded into the bare
9
+ # domain, and anything that isn't an http(s) URL with a host
10
+ # (`chrome://`, `file:`, `about:`, `javascript:`) is dropped.
11
+ module Domains
12
+ # @return [Hash{String => Hash}] domain => { hosts: {host => visits},
13
+ # rows: [...], visits: total }.
14
+ def self.group(rows)
15
+ groups = {}
16
+ rows.each do |row|
17
+ host = host_of(row[:url])
18
+ next unless host
19
+
20
+ group = (groups[host.delete_prefix("www.")] ||= { hosts: Hash.new(0), rows: [], visits: 0 })
21
+ group[:hosts][host] += row[:visits]
22
+ group[:rows] << row
23
+ group[:visits] += row[:visits]
24
+ end
25
+ groups
26
+ end
27
+
28
+ # The most-visited host variant (www. or not) as an https origin —
29
+ # UCP is https-only, so an http-only visit still probes https.
30
+ def self.origin_for(group) = "https://#{group[:hosts].max_by { |_h, visits| visits }.first}"
31
+
32
+ # An index entry's origin, reduced the same way, so it compares
33
+ # against a group's key.
34
+ def self.key_of(origin) = host_of(origin)&.delete_prefix("www.")
35
+
36
+ def self.host_of(url)
37
+ uri = URI.parse(url.to_s)
38
+ return nil unless %w[http https].include?(uri.scheme) && uri.host && !uri.host.empty?
39
+
40
+ uri.host.downcase
41
+ rescue URI::InvalidURIError
42
+ nil
43
+ end
44
+ end
45
+ end
46
+ end
47
+ end
@@ -0,0 +1,91 @@
1
+ require "ipaddr"
2
+
3
+ require_relative "../search_backends"
4
+
5
+ module Portage
6
+ module Cli
7
+ module BrowserImport
8
+ # "Is this host obviously not a shop?" — decided locally, before a
9
+ # single probe is spent (docs/plans/buy-skill-and-local-browser.md
10
+ # Phase 3: "skipping obvious non-shops (NON_STORE_HOSTS, webmail,
11
+ # banks, intranet and localhost hosts)"). A skipped host never leaves
12
+ # the machine at all, so these lists err on the side of skipping:
13
+ # a missed shop costs one `portage index add`, a probed bank costs
14
+ # the user's trust.
15
+ module Filter
16
+ WEBMAIL_HOSTS = %w[
17
+ mail.google.com outlook.live.com outlook.office.com outlook.office365.com mail.yahoo.com
18
+ proton.me protonmail.com icloud.com fastmail.com mail.aol.com gmx.com zoho.com
19
+ ].freeze
20
+
21
+ # Banks and payment/finance hosts are never a UCP storefront, and
22
+ # are exactly the hosts a shopper least wants anything sent about.
23
+ # Not exhaustive — backed up by BANK_WORD for the long tail.
24
+ BANK_HOSTS = %w[
25
+ paypal.com wise.com revolut.com monzo.com starlingbank.com chase.com wellsfargo.com
26
+ citi.com capitalone.com americanexpress.com discover.com schwab.com fidelity.com vanguard.com
27
+ hsbc.com hsbc.co.uk barclays.co.uk natwest.com santander.co.uk nationwide.co.uk halifax.co.uk
28
+ coinbase.com kraken.com stripe.com klarna.com
29
+ ].freeze
30
+
31
+ BANK_WORD = /bank|creditunion/
32
+
33
+ # Tools, social and media sites a shopper visits all day that will
34
+ # never answer /.well-known/ucp — skipping them saves probes for
35
+ # the domains that might.
36
+ NON_SHOP_HOSTS = %w[
37
+ github.com gitlab.com bitbucket.org stackoverflow.com stackexchange.com linkedin.com slack.com
38
+ zoom.us notion.so atlassian.net figma.com claude.ai anthropic.com openai.com chatgpt.com
39
+ instagram.com tiktok.com netflix.com spotify.com twitch.tv discord.com whatsapp.com
40
+ medium.com substack.com apple.com microsoft.com live.com office.com gstatic.com
41
+ googleusercontent.com cloudflare.com amazonaws.com
42
+ ].freeze
43
+
44
+ LOCAL_SUFFIXES = %w[localhost local internal lan corp home.arpa test intranet].freeze
45
+
46
+ # @return [Symbol, nil] why `host` is skipped (:non_store, :webmail,
47
+ # :bank, :local), or nil when it's worth a probe.
48
+ def self.skip_reason(host)
49
+ host = host.to_s.downcase
50
+ return :local if local?(host)
51
+ return :webmail if webmail?(host)
52
+ return :bank if bank?(host)
53
+ return :non_store if non_store?(host)
54
+
55
+ nil
56
+ end
57
+
58
+ def self.local?(host)
59
+ return true if host.empty? || !host.include?(".")
60
+ return true if ip?(host)
61
+
62
+ LOCAL_SUFFIXES.any? { |suffix| host == suffix || host.end_with?(".#{suffix}") }
63
+ end
64
+
65
+ def self.webmail?(host)
66
+ host.start_with?("mail.", "webmail.") || matches?(host, WEBMAIL_HOSTS)
67
+ end
68
+
69
+ def self.bank?(host)
70
+ matches?(host, BANK_HOSTS) || host.match?(BANK_WORD)
71
+ end
72
+
73
+ # SearchBackends::NON_STORE_HOSTS (search engines, Wikipedia,
74
+ # social) plus this module's own NON_SHOP_HOSTS.
75
+ def self.non_store?(host)
76
+ !SearchBackends.store_candidate?("https://#{host}/") || matches?(host, NON_SHOP_HOSTS)
77
+ end
78
+
79
+ def self.matches?(host, list) = list.any? { |bad| host == bad || host.end_with?(".#{bad}") }
80
+
81
+ def self.ip?(host)
82
+ IPAddr.new(host.delete_prefix("[").delete_suffix("]"))
83
+ true
84
+ rescue IPAddr::InvalidAddressError, IPAddr::AddressFamilyError
85
+ false
86
+ end
87
+ private_class_method :local?, :webmail?, :bank?, :non_store?, :matches?, :ip?
88
+ end
89
+ end
90
+ end
91
+ end
@@ -0,0 +1,248 @@
1
+ require "uri"
2
+
3
+ require_relative "../classifier"
4
+ require_relative "../index"
5
+ require_relative "../index/builder"
6
+ require_relative "profiles"
7
+ require_relative "readers"
8
+ require_relative "filter"
9
+ require_relative "prober"
10
+ require_relative "categorize"
11
+ require_relative "saver"
12
+ require_relative "domains"
13
+
14
+ module Portage
15
+ module Cli
16
+ module BrowserImport
17
+ # `portage browser import` (docs/plans/buy-skill-and-local-browser.md
18
+ # Phase 3, Tier A): bookmarks and history in, a short list of shop
19
+ # domains out, written to the user's own Index::Store only after the
20
+ # user has seen the list and said yes (Cli.run_browser_import owns
21
+ # that gate; this class only ever writes from #save).
22
+ #
23
+ # #plan reads the profile's allowed files (Profiles::ALLOWED_FILES,
24
+ # nothing else), reduces every row to its domain, and decides each
25
+ # domain locally first — skipped as an obvious non-shop (Filter),
26
+ # excluded by the user, on the hand-off-only list, or already in the
27
+ # index — before spending one of at most `max_probes` Prober probes
28
+ # on an unknown one. A domain is kept only when it answers
29
+ # `/.well-known/ucp`, matches a WebMCP preset, or is hand-off only.
30
+ # Each kept domain is classified from what the user actually looked
31
+ # at there (page titles, bookmark folder names, URL slugs), weighted
32
+ # by visit count.
33
+ #
34
+ # Kept domains land in Index::Store with `sources: ["history"]`/
35
+ # `["bookmark"]` — the same untrusted index every other source
36
+ # writes, so an imported store reaches `find` only as a
37
+ # `source: "index"` candidate (never Policy#merchant_allowlist, never
38
+ # past the `--store`/interactive pick a `--yes` buy requires), and
39
+ # Index::Exporter treats both labels as personal: never exported.
40
+ class Importer
41
+ DEFAULT_HISTORY_DAYS = 90
42
+ MAX_PROBES = 200
43
+
44
+ FULL_DISK_ACCESS = "macOS won't let this terminal read Safari's history and bookmarks without Full Disk " \
45
+ "Access. To allow it, open System Settings > Privacy & Security > Full Disk Access, " \
46
+ "turn it on for the terminal app you run portage from, then quit and reopen that app " \
47
+ "and re-run this command. Portage doesn't try any other way in. Nothing was imported, " \
48
+ "probed or saved.".freeze
49
+
50
+ # Any other browser: the same "explain and stop", without claiming
51
+ # to know which protection said no (macOS privacy settings, a
52
+ # sandboxed terminal, file permissions).
53
+ PERMISSION_DENIED = "Permission denied reading %<browser>s's profile folder. If your terminal runs " \
54
+ "sandboxed or macOS is protecting that folder, allow access (System Settings > " \
55
+ "Privacy & Security > Full Disk Access) and re-run. Portage doesn't try any other way " \
56
+ "in. Nothing was imported, probed or saved.".freeze
57
+
58
+ Options = Struct.new(:browser, :root, :history_days, :include_product_pages, :max_probes, :exclude,
59
+ keyword_init: true) do
60
+ def initialize(history_days: DEFAULT_HISTORY_DAYS, include_product_pages: false, max_probes: MAX_PROBES,
61
+ exclude: [], **)
62
+ super
63
+ end
64
+ end
65
+
66
+ # @param handoff_only_hosts [Array<String>] Tier C hosts (Phase 5
67
+ # will supply the user's own `handoff_only_hosts:` config; nothing
68
+ # ships that list yet, so it defaults to empty). A history domain
69
+ # on it is kept — hand-off only, never probed, never automated.
70
+ # @param webmcp_preset_for [#call, nil] `->(origin) { preset_or_nil }`.
71
+ # Matching a WebMCP preset needs a browser bridge an import doesn't
72
+ # have (same posture as Index::Sources::WebmcpSweep), so nil (the
73
+ # default) skips the check entirely.
74
+ # @param readers [#call] `->(family) { reader }` — see Readers.for.
75
+ def initialize(stores: Index::Store.new, products: Index::ProductStore.new,
76
+ known_cache: Index::KnownCache.new, prober: Prober.new, handoff_only_hosts: [],
77
+ webmcp_preset_for: nil, readers: Readers.method(:for), now: Time.now)
78
+ @stores = stores
79
+ @products = products
80
+ @known_cache = known_cache
81
+ @prober = prober
82
+ @handoff_only_hosts = handoff_only_hosts.map { |h| h.to_s.downcase.delete_prefix("www.") }
83
+ @webmcp_preset_for = webmcp_preset_for
84
+ @readers = readers
85
+ @now = now
86
+ end
87
+
88
+ # @return [Hash] the proposal: counts, `kept` (domains to save) and
89
+ # `products` (only with include_product_pages) — or `error:` /
90
+ # `message:` when the browser's files couldn't be read (Safari
91
+ # without Full Disk Access, no sqlite3, no profile found).
92
+ def plan(options)
93
+ profiles = Profiles.locate(options.browser, root: options.root)
94
+ return no_profile(options) if profiles.empty?
95
+
96
+ rows, opened = read_rows(profiles, since: @now.to_i - (options.history_days.to_i * 86_400))
97
+ summarize(options, profiles, opened, rows)
98
+ rescue Sqlite::PermissionDenied
99
+ permission_denied(options)
100
+ rescue Sqlite::Unavailable => e
101
+ { browser: options.browser, error: "reader_unavailable",
102
+ message: "Couldn't read the browser's files: #{e.message}." }
103
+ end
104
+
105
+ # Writes a #plan's kept domains (and products) into the index — see
106
+ # Saver.
107
+ # @return [Hash] stores:, products: — how many entries were written.
108
+ def save(proposal) = Saver.new(stores: @stores, products: @products, now: @now).save(proposal)
109
+
110
+ private
111
+
112
+ # --- Reading ---
113
+
114
+ def read_rows(profiles, since:)
115
+ opened = []
116
+ rows = profiles.flat_map do |profile|
117
+ reader = @readers.call(profile.family)
118
+ opened.concat([profile.history, profile.bookmarks].compact)
119
+ reader.history(profile, since: since) + reader.bookmarks(profile)
120
+ end
121
+ [rows, opened.uniq]
122
+ end
123
+
124
+ # --- Deciding each domain ---
125
+
126
+ def summarize(options, profiles, opened, rows)
127
+ groups = Domains.group(rows)
128
+ state = { skipped: Hash.new(0), kept: [], cached_miss: 0, not_ucp: 0, unprobed: 0 }
129
+ groups.sort_by { |_key, g| -g[:visits] }.each { |key, group| decide(key, group, options, state) }
130
+ report(options, profiles, opened, rows, groups, state)
131
+ end
132
+
133
+ def decide(key, group, options, state)
134
+ reason = local_skip_reason(key, options)
135
+ return state[:skipped][reason] += 1 if reason
136
+
137
+ verdict = local_verdict(key) || probe_verdict(Domains.origin_for(group), options, state)
138
+ state[:kept] << kept_entry(key, group, verdict) if verdict
139
+ end
140
+
141
+ def local_skip_reason(key, options)
142
+ return :excluded if Array(options.exclude).any? { |h| h.to_s.downcase.delete_prefix("www.") == key }
143
+
144
+ Filter.skip_reason(key)
145
+ end
146
+
147
+ # Everything decidable with no network call at all.
148
+ def local_verdict(key)
149
+ return { verdict: "handoff_only", handoff_only: true } if handoff_only?(key)
150
+
151
+ own = own_entry(key)
152
+ return { verdict: "indexed", origin: own["origin"] } if own
153
+
154
+ known = known_entry(key)
155
+ return unless known
156
+
157
+ { verdict: "known", origin: known["origin"], capabilities: known["capabilities"],
158
+ last_verified: known["last_verified"] }
159
+ end
160
+
161
+ # A cached "no UCP" verdict or the probe cap skips the probe, but
162
+ # not the WebMCP check (when a bridge is injected) — a page can
163
+ # expose WebMCP tools without any manifest at all.
164
+ def probe_verdict(origin, options, state)
165
+ session = ucp_session(origin, options, state)
166
+ return nil if session == :capped
167
+ return { verdict: "ucp", origin: origin, capabilities: Index::Builder.capabilities_of(session) } if session
168
+
169
+ preset = @webmcp_preset_for&.call(origin)
170
+ return { verdict: "webmcp", origin: origin, webmcp_preset: preset } if preset
171
+
172
+ state[:not_ucp] += 1 unless session == false
173
+ nil
174
+ end
175
+
176
+ # @return [Session, nil, false, :capped] a session; nil when the
177
+ # probe found no UCP; false for a cached miss (no request made);
178
+ # :capped once max_probes is spent.
179
+ def ucp_session(origin, options, state)
180
+ if @prober.cached_miss?(origin)
181
+ state[:cached_miss] += 1
182
+ false
183
+ elsif @prober.probes >= options.max_probes.to_i
184
+ state[:unprobed] += 1
185
+ :capped
186
+ else
187
+ @prober.probe(origin)
188
+ end
189
+ end
190
+
191
+ def handoff_only?(key)
192
+ @handoff_only_hosts.any? { |h| key == h || key.end_with?(".#{h}") }
193
+ end
194
+
195
+ def own_entry(key) = @stores.all.find { |e| Domains.key_of(e["origin"]) == key }
196
+
197
+ def known_entry(key) = @known_cache.stores.values.find { |e| Domains.key_of(e["origin"]) == key }
198
+
199
+ # --- The proposal ---
200
+
201
+ def kept_entry(key, group, verdict)
202
+ { domain: key, origin: verdict[:origin] || Domains.origin_for(group), verdict: verdict[:verdict],
203
+ sources: group[:rows].map { |r| r[:kind] }.uniq.sort, visits: group[:visits],
204
+ categories: Categorize.domain(group[:rows]), rows: group[:rows] }
205
+ .merge(verdict.slice(:capabilities, :webmcp_preset, :handoff_only, :last_verified))
206
+ end
207
+
208
+ def report(options, profiles, opened, rows, groups, state)
209
+ kept = state[:kept]
210
+ { browser: options.browser, profiles: profiles.length, files_opened: opened, rows: row_counts(rows),
211
+ domains: groups.length, skipped: state[:skipped].transform_keys(&:to_s),
212
+ already_indexed: count(kept, "indexed"), known: count(kept, "known"), probed: @prober.probes,
213
+ cached_miss: state[:cached_miss], not_ucp: state[:not_ucp], unprobed: state[:unprobed],
214
+ capped: state[:unprobed].positive?, kept: kept.map { |entry| public_entry(entry) },
215
+ products: options.include_product_pages ? Categorize.products(kept) : [] }
216
+ end
217
+
218
+ def row_counts(rows)
219
+ kinds = rows.map { |r| r[:kind] }
220
+ { history: kinds.count("history"), bookmark: kinds.count("bookmark") }
221
+ end
222
+
223
+ def public_entry(entry)
224
+ entry.except(:rows).merge(category_names: Classifier.names_for(entry[:categories].keys))
225
+ end
226
+
227
+ def count(kept, verdict) = kept.count { |e| e[:verdict] == verdict }
228
+
229
+ # --- Helpers ---
230
+
231
+ def no_profile(options)
232
+ where = options.root ? " under #{options.root}" : ""
233
+ { browser: options.browser, error: "no_profile",
234
+ message: "No #{options.browser} profile with history or bookmarks found#{where}." }
235
+ end
236
+
237
+ def permission_denied(options)
238
+ if options.browser == "safari"
239
+ { browser: "safari", error: "full_disk_access_required", message: FULL_DISK_ACCESS }
240
+ else
241
+ { browser: options.browser, error: "permission_denied",
242
+ message: format(PERMISSION_DENIED, browser: options.browser) }
243
+ end
244
+ end
245
+ end
246
+ end
247
+ end
248
+ end
@@ -0,0 +1,72 @@
1
+ require "strscan"
2
+
3
+ module Portage
4
+ module Cli
5
+ module BrowserImport
6
+ # Just enough of Apple's XML property-list format to read Safari's
7
+ # Bookmarks.plist once `plutil -convert xml1` has turned its binary
8
+ # form into XML. `plutil -convert json` would be simpler but refuses
9
+ # any plist holding a <date> (Safari's Reading List entries carry
10
+ # them), and a real XML library (rexml) isn't a runtime dependency of
11
+ # this gem — this is ~50 lines instead of a new one.
12
+ #
13
+ # Dates, data and numbers come back as their raw text; bookmark
14
+ # import only ever reads strings out of the tree.
15
+ module PlistXml
16
+ class ParseError < StandardError; end
17
+
18
+ SCALARS = "string|key|data|date|integer|real".freeze
19
+ ENTITIES = { "amp" => "&", "lt" => "<", "gt" => ">", "quot" => '"', "apos" => "'" }.freeze
20
+
21
+ def self.parse(xml)
22
+ scanner = StringScanner.new(xml.to_s)
23
+ raise ParseError, "not an XML plist" unless scanner.skip_until(/<plist[^>]*>/)
24
+
25
+ value(scanner)
26
+ end
27
+
28
+ def self.value(scanner)
29
+ scanner.skip(/\s*/)
30
+ if scanner.scan(%r{<(dict|array|#{SCALARS})\s*/>}) then empty(scanner[1])
31
+ elsif scanner.scan(%r{<(true|false)\s*/>}) then scanner[1] == "true"
32
+ elsif scanner.scan("<dict>") then dict(scanner)
33
+ elsif scanner.scan("<array>") then array(scanner)
34
+ elsif scanner.scan(%r{<(#{SCALARS})>(.*?)</\1>}m) then unescape(scanner[2])
35
+ else raise ParseError, "unexpected plist content at #{scanner.pos}"
36
+ end
37
+ end
38
+
39
+ def self.dict(scanner)
40
+ result = {}
41
+ until scanner.skip(%r{\s*</dict>})
42
+ key = value(scanner)
43
+ raise ParseError, "dict key isn't a string" unless key.is_a?(String)
44
+
45
+ result[key] = value(scanner)
46
+ end
47
+ result
48
+ end
49
+
50
+ def self.array(scanner)
51
+ result = []
52
+ result << value(scanner) until scanner.skip(%r{\s*</array>})
53
+ result
54
+ end
55
+
56
+ def self.empty(tag)
57
+ { "dict" => {}, "array" => [] }.fetch(tag, "")
58
+ end
59
+
60
+ def self.unescape(text)
61
+ text.gsub(/&(#x?)?(\w+);/) do
62
+ if Regexp.last_match(1) == "#x" then [Regexp.last_match(2).to_i(16)].pack("U")
63
+ elsif Regexp.last_match(1) == "#" then [Regexp.last_match(2).to_i].pack("U")
64
+ else ENTITIES.fetch(Regexp.last_match(2), Regexp.last_match(0))
65
+ end
66
+ end
67
+ end
68
+ private_class_method :value, :dict, :array, :empty, :unescape
69
+ end
70
+ end
71
+ end
72
+ end
@@ -0,0 +1,60 @@
1
+ require "timeout"
2
+ require "portage/ucp/client"
3
+
4
+ require_relative "../user_agent"
5
+ require_relative "../probe_cache"
6
+
7
+ module Portage
8
+ module Cli
9
+ module BrowserImport
10
+ # The only way a browser-imported domain ever leaves this machine:
11
+ # one `GET /.well-known/ucp` through Portage::Ucp::Client.discover
12
+ # (the same discovery path `find` and `index build` use), recorded
13
+ # in the shared ProbeCache so a domain that didn't answer last time
14
+ # isn't asked again. No page, title, path or visit count is ever
15
+ # sent — just the origin, as the host of that one request.
16
+ class Prober
17
+ THROTTLE = 0.1
18
+
19
+ # Net::HTTP's own defaults are 60s each way — fine for one store,
20
+ # not for 200 history domains, some of which will be dead hosts.
21
+ TIMEOUT = 5
22
+
23
+ # @param discover [#call, nil] `->(origin) { session_or_nil }` —
24
+ # injectable so specs never make a real request; nil (the
25
+ # default) uses Portage::Ucp::Client.discover.
26
+ def initialize(cache: ProbeCache.new, discover: nil, throttle: THROTTLE, timeout: TIMEOUT)
27
+ @cache = cache
28
+ @discover = discover || method(:ucp_discover)
29
+ @throttle = throttle
30
+ @timeout = timeout
31
+ @probes = 0
32
+ end
33
+
34
+ attr_reader :probes
35
+
36
+ # A cached "no UCP here" verdict — answered with no request at all.
37
+ def cached_miss?(origin) = @cache.fetch(origin) == false
38
+
39
+ # @return [Portage::Ucp::Client::Session, nil]
40
+ def probe(origin)
41
+ sleep(@throttle) if @probes.positive? && @throttle.to_f.positive?
42
+ @probes += 1
43
+ session = safely { @discover.call(origin) }
44
+ @cache.record(origin, !session.nil?)
45
+ session
46
+ end
47
+
48
+ private
49
+
50
+ def safely(&)
51
+ Timeout.timeout(@timeout, &)
52
+ rescue StandardError # Timeout::Error included
53
+ nil
54
+ end
55
+
56
+ def ucp_discover(origin) = Portage::Ucp::Client.discover(origin, headers: UserAgent.headers)
57
+ end
58
+ end
59
+ end
60
+ end