sbuilder-mcp 0.16.1 → 0.17.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +17 -0
- package/CHANGELOG.vi.md +17 -0
- package/README.md +1 -0
- package/README.vi.md +1 -0
- package/dist/core/patch.js +32 -2
- package/dist/domains/site/discover.js +347 -0
- package/dist/domains/site/traps.js +27 -0
- package/dist/tools/importpage.js +429 -17
- package/dist/tools/page.js +12 -0
- package/dist/vision/capture.js +194 -19
- package/package.json +1 -1
package/CHANGELOG.md
CHANGED
|
@@ -6,6 +6,23 @@ All notable changes to this project are documented in this file.
|
|
|
6
6
|
The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
|
|
7
7
|
and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
|
8
8
|
|
|
9
|
+
## [0.17.1] - 2026-09-10
|
|
10
|
+
|
|
11
|
+
### Fixed
|
|
12
|
+
- sb_add and sb_import no longer store a nested node's children twice: a patch batch that gets applied more than once (as every write already is, to validate it before it lands) mutated itself on the first pass by carrying an added node by reference, so the second pass re-inserted its children into a node that already held them.
|
|
13
|
+
- sb_import and sb_import_site now cap a flattened row at 12 columns and treat a grid container as wrapping rather than single-line, so a page whose content wrapper is a CSS grid (a documentation site, for example) no longer imports as one row of hundreds of slivered columns.
|
|
14
|
+
- sb_import and sb_import_site now capture a `<pre>`/`<code>` block as one node instead of one node per syntax-highlighting `<span>`, so a code sample no longer arrives broken into dozens of single-token fragments.
|
|
15
|
+
|
|
16
|
+
## [0.17.0] - 2026-09-10
|
|
17
|
+
|
|
18
|
+
### Added
|
|
19
|
+
- sb_import_site reads a whole site from one URL — the publisher's own sitemap first, a bounded link crawl only when there is none — and creates and fills a draft page here for every page it finds, replacing the by-hand loop of one sb_import call per page.
|
|
20
|
+
- sb_import_site reports every repeated path prefix (such as `/products/{slug}`) as one bound template before creating anything, since importing those URLs as static pages would produce a catalogue where nothing is buyable.
|
|
21
|
+
- sb_import_site previews where each page will land, including which URL merges into the site's existing home page and which slug is already taken, and skips a page whose slug collides instead of letting the platform silently rename it.
|
|
22
|
+
|
|
23
|
+
### Fixed
|
|
24
|
+
- sb_import and sb_import_site now insert imported content into the middle band, before the first global footer, instead of appending it to the end of the page, since appending broke the platform's band-order rule on every page carrying a global footer.
|
|
25
|
+
|
|
9
26
|
## [0.16.1] - 2026-09-10
|
|
10
27
|
|
|
11
28
|
### Fixed
|
package/CHANGELOG.vi.md
CHANGED
|
@@ -6,6 +6,23 @@ Mọi thay đổi đáng chú ý của dự án được ghi lại trong file n
|
|
|
6
6
|
Định dạng dựa trên [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
|
|
7
7
|
và dự án tuân theo [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
|
8
8
|
|
|
9
|
+
## [0.17.1] - 2026-09-10
|
|
10
|
+
|
|
11
|
+
### Fixed
|
|
12
|
+
- sb_add và sb_import giờ không còn lưu nhân đôi các con của một node lồng nhau: một batch patch bị áp dụng nhiều hơn một lần (như mọi lần ghi vẫn vậy, để kiểm tra trước khi áp thật) trước đây tự làm thay đổi chính nó ngay ở lượt đầu vì mang theo một node vừa thêm bằng tham chiếu, khiến lượt thứ hai chèn lại các con vào một node đã sẵn có chúng.
|
|
13
|
+
- sb_import và sb_import_site giờ giới hạn một row được làm phẳng ở tối đa 12 cột và coi container dạng grid là tự xuống dòng thay vì nằm trên một hàng, để một trang có content wrapper là CSS grid (chẳng hạn một trang tài liệu kỹ thuật) không còn bị import thành một row hàng trăm cột mỏng dính.
|
|
14
|
+
- sb_import và sb_import_site giờ lấy một khối `<pre>`/`<code>` thành một node duy nhất thay vì một node cho mỗi `<span>` tô cú pháp, để một đoạn code không còn bị vỡ thành hàng chục mảnh một-token-một-node.
|
|
15
|
+
|
|
16
|
+
## [0.17.0] - 2026-09-10
|
|
17
|
+
|
|
18
|
+
### Added
|
|
19
|
+
- sb_import_site giờ đọc toàn bộ một site từ một URL — ưu tiên sitemap của chính nhà xuất bản, chỉ crawl link có giới hạn khi không có sitemap — rồi tạo và điền một draft page ở đây cho từng trang tìm được, thay cho vòng lặp thủ công gọi sb_import từng trang một.
|
|
20
|
+
- sb_import_site giờ báo cáo mọi tiền tố đường dẫn lặp lại (như `/products/{slug}`) là một template gắn với nền tảng trước khi tạo bất cứ gì, vì import các URL đó thành trang tĩnh sẽ tạo ra một catalogue mà không gì có thể mua được.
|
|
21
|
+
- sb_import_site giờ xem trước từng trang sẽ nằm ở đâu, gồm cả URL nào sẽ gộp vào trang chủ hiện có của site và slug nào đã bị chiếm, rồi bỏ qua trang có slug trùng thay vì để nền tảng âm thầm đổi tên nó.
|
|
22
|
+
|
|
23
|
+
### Fixed
|
|
24
|
+
- sb_import và sb_import_site giờ chèn nội dung import vào giữa các band, trước global footer đầu tiên, thay vì nối vào cuối trang, vì việc nối vào cuối trước đây phá vỡ quy tắc thứ tự band của nền tảng trên mọi trang có global footer.
|
|
25
|
+
|
|
9
26
|
## [0.16.1] - 2026-09-10
|
|
10
27
|
|
|
11
28
|
### Fixed
|
package/README.md
CHANGED
|
@@ -99,6 +99,7 @@ make, because those mean "this person's account".
|
|
|
99
99
|
| `sb_event` | Give a node a click action — open the cart, go to a page, open a pop-up |
|
|
100
100
|
| `sb_bind` | Bind a node's content to real store data, or make a button add to the cart |
|
|
101
101
|
| `sb_import` | Read a page from any public URL and add its structure and content to the open page as real elements, styled with THIS page's own tokens — a translation, not a clone |
|
|
102
|
+
| `sb_import_site` | Read a WHOLE site from one URL — its sitemap, or the links on that page — and give each page found its own draft page here, built from this site's tokens |
|
|
102
103
|
| `sb_store` | Run a store flow that must happen in a fixed order — the four writes that make a working checkout, or any of the platform's 17 form templates (login, register, forgot, contact, subscribe …) with its own field document |
|
|
103
104
|
| `sb_undo` | Put back what a PUT replaced — the platform has no page history or restore, so this is the only way back |
|
|
104
105
|
|
package/README.vi.md
CHANGED
|
@@ -96,6 +96,7 @@ là "tài khoản của người này".
|
|
|
96
96
|
| `sb_event` | Gắn click action cho một node — mở giỏ, sang trang, mở pop-up |
|
|
97
97
|
| `sb_bind` | Gắn nội dung một node vào dữ liệu cửa hàng thật, hoặc biến một nút thành nút thêm vào giỏ |
|
|
98
98
|
| `sb_import` | Đọc một trang từ URL công khai bất kỳ và thêm cấu trúc + nội dung của nó vào trang đang mở dưới dạng element thật, mang token của CHÍNH trang này — là dịch lại, không phải sao chép |
|
|
99
|
+
| `sb_import_site` | Đọc CẢ website từ một URL — sitemap của nó, hoặc các link trên trang đó — và tạo cho mỗi trang tìm được một trang nháp riêng ở đây, dựng bằng token của site này |
|
|
99
100
|
| `sb_store` | Chạy một luồng cửa hàng bắt buộc đúng thứ tự — bốn lệnh ghi tạo nên trang thanh toán, hoặc gieo bất kỳ template nào trong 17 form của nền tảng (login, register, forgot, contact, subscribe …) kèm field document của nó |
|
|
100
101
|
| `sb_undo` | Trả lại thứ mà một lệnh PUT đã ghi đè — nền tảng không có lịch sử trang hay restore, nên đây là đường về duy nhất |
|
|
101
102
|
|
package/dist/core/patch.js
CHANGED
|
@@ -78,6 +78,16 @@ function parentOf(state, path) {
|
|
|
78
78
|
* half-edited with nothing anywhere to say so — which is the exact failure shape
|
|
79
79
|
* this whole repo exists to rule out.
|
|
80
80
|
*/
|
|
81
|
+
/**
|
|
82
|
+
* A patch's value, detached from whatever the caller still holds.
|
|
83
|
+
*
|
|
84
|
+
* Primitives are returned as they are — a style key is a string, and cloning
|
|
85
|
+
* one on every `sb_set` would be pure cost. Only a structure can be aliased,
|
|
86
|
+
* and only an alias can be mutated behind the batch's back.
|
|
87
|
+
*/
|
|
88
|
+
function clone(value) {
|
|
89
|
+
return value !== null && typeof value === 'object' ? structuredClone(value) : value;
|
|
90
|
+
}
|
|
81
91
|
export function applyPatches(state, patches) {
|
|
82
92
|
for (const p of patches) {
|
|
83
93
|
if (!isSyncablePatch(p)) {
|
|
@@ -86,7 +96,27 @@ export function applyPatches(state, patches) {
|
|
|
86
96
|
const { holder, key } = parentOf(state, p.path);
|
|
87
97
|
switch (p.op) {
|
|
88
98
|
case 'set':
|
|
89
|
-
|
|
99
|
+
// THE VALUE IS NEVER ALIASED INTO THE DOCUMENT.
|
|
100
|
+
//
|
|
101
|
+
// `addSubtree` builds a node, emits `set nodes/<id>` carrying THAT
|
|
102
|
+
// object, and then emits `insert` patches that push child ids into
|
|
103
|
+
// `data.nodes`. Assigning the reference means the first apply MUTATES
|
|
104
|
+
// the patch's own value, so the batch is no longer the thing it was: a
|
|
105
|
+
// second apply re-establishes a node that already holds its children and
|
|
106
|
+
// then inserts them again.
|
|
107
|
+
//
|
|
108
|
+
// Applying a batch twice is not hypothetical — `applyAndSave` does it by
|
|
109
|
+
// design, once through `preview` to judge the write and once for real.
|
|
110
|
+
// Between v0.16.1 (which introduced that check) and this fix, EVERY
|
|
111
|
+
// nested `sb_add` stored each child twice, and `sb_import` three times.
|
|
112
|
+
// It is silent: the tree is well-formed, every id resolves, the save is
|
|
113
|
+
// accepted, and the page simply renders its content twice. Measured on
|
|
114
|
+
// a real import — 106 of 231 containers listing one child id three
|
|
115
|
+
// times.
|
|
116
|
+
//
|
|
117
|
+
// A copy makes the batch idempotent for this shape: the re-`set` puts a
|
|
118
|
+
// pristine node back, and the inserts that follow rebuild the same list.
|
|
119
|
+
holder[key] = clone(p.value);
|
|
90
120
|
break;
|
|
91
121
|
case 'unset':
|
|
92
122
|
delete holder[key];
|
|
@@ -95,7 +125,7 @@ export function applyPatches(state, patches) {
|
|
|
95
125
|
const arr = holder[key];
|
|
96
126
|
if (!Array.isArray(arr))
|
|
97
127
|
break;
|
|
98
|
-
arr.splice(p.index, 0, p.value);
|
|
128
|
+
arr.splice(p.index, 0, clone(p.value));
|
|
99
129
|
break;
|
|
100
130
|
}
|
|
101
131
|
case 'remove': {
|
|
@@ -0,0 +1,347 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* WHICH PAGES A SITE HAS, starting from one URL.
|
|
3
|
+
*
|
|
4
|
+
* `sb_import` reads ONE page, and that is the right shape for "bring this
|
|
5
|
+
* section over". It is the wrong shape for the thing people actually ask for —
|
|
6
|
+
* hand over a link and get the site — because the missing half was never the
|
|
7
|
+
* capture: it was knowing what the pages ARE, and creating one for each.
|
|
8
|
+
*
|
|
9
|
+
* Two sources, in that order, because they cost wildly different amounts:
|
|
10
|
+
*
|
|
11
|
+
* 1. The site's own sitemap. One HTTP fetch, no browser, and it is the
|
|
12
|
+
* publisher's own list rather than a guess — a page nothing links to is in
|
|
13
|
+
* it, and a link that goes nowhere is not.
|
|
14
|
+
* 2. The links on the entry page. A browser navigation each, so it is bounded
|
|
15
|
+
* hard, and it is the fallback rather than the default for that reason.
|
|
16
|
+
*
|
|
17
|
+
* This module is PURE, for the same reason `importmap.ts` is: the rules about
|
|
18
|
+
* what counts as a page are the half worth arguing over, and they can be argued
|
|
19
|
+
* over offline. Fetching belongs to the caller.
|
|
20
|
+
*/
|
|
21
|
+
/** Anything whose extension says it is a file rather than a page. */
|
|
22
|
+
const ASSET = /\.(?:pdf|jpe?g|png|gif|webp|svg|ico|avif|css|js|mjs|json|xml|rss|atom|zip|rar|gz|tgz|tar|mp3|mp4|m4a|webm|mov|avi|woff2?|ttf|otf|eot|docx?|xlsx?|pptx?|csv|txt)$/i;
|
|
23
|
+
/**
|
|
24
|
+
* Paths that are a SITE'S PLUMBING rather than its content.
|
|
25
|
+
*
|
|
26
|
+
* Every one of these is either somebody else's storefront machinery — a cart, a
|
|
27
|
+
* login, an account page this platform serves at its own fixed path — or an
|
|
28
|
+
* endless tail (`/tag/`, `/author/`) that would spend the whole page budget on
|
|
29
|
+
* near-duplicates of pages already taken. A caller who genuinely wants one says
|
|
30
|
+
* so through `include`, which is checked first.
|
|
31
|
+
*/
|
|
32
|
+
const NOT_CONTENT = [
|
|
33
|
+
'/wp-admin', '/wp-login', '/wp-json', '/wp-content', '/xmlrpc', '/cdn-cgi',
|
|
34
|
+
'/feed', '/rss', '/comments',
|
|
35
|
+
'/cart', '/checkout', '/my-account', '/account', '/login', '/logout',
|
|
36
|
+
'/register', '/signin', '/sign-in', '/signup', '/sign-up', '/password',
|
|
37
|
+
'/search', '/wishlist', '/compare',
|
|
38
|
+
// Trailing slash means "this segment, wherever it appears": a tag archive is
|
|
39
|
+
// as often /blog/tag/x as /tag/x.
|
|
40
|
+
'/tag/', '/tags/', '/author/', '/authors/',
|
|
41
|
+
];
|
|
42
|
+
/**
|
|
43
|
+
* Is this path the site's plumbing rather than one of its pages?
|
|
44
|
+
*
|
|
45
|
+
* A SUBSTRING TEST IS THE WRONG TEST, and it drops real pages: `/feedback`
|
|
46
|
+
* contains `/feed`, `/cartier-watches` contains `/cart`, `/comments-policy`
|
|
47
|
+
* contains `/comments`. Each of those is an ordinary page a merchant would
|
|
48
|
+
* expect to see imported, and the loss is reported only as a number.
|
|
49
|
+
*
|
|
50
|
+
* So a plain entry matches at a BOUNDARY — the end of the path, a `/`, or a `.`
|
|
51
|
+
* so `/wp-login.php` is still caught — and an entry written with a trailing
|
|
52
|
+
* slash matches that segment anywhere.
|
|
53
|
+
*
|
|
54
|
+
* ANCHORING IT AT THE START OF THE PATH WAS THE FIRST FIX AND IT WAS HALF ONE.
|
|
55
|
+
* A locale prefix is the ordinary shape of the sites this tool is pointed at,
|
|
56
|
+
* and this platform's own market is Vietnamese, so `/en/cart`, `/vi/account` and
|
|
57
|
+
* `/shop/checkout` all sailed through — each one eating a page slot and handing
|
|
58
|
+
* the merchant a junk draft. The needle already begins with `/`, so its own left
|
|
59
|
+
* boundary comes free: scanning anywhere and testing only the RIGHT boundary
|
|
60
|
+
* catches those without reopening `/cartier-watches`.
|
|
61
|
+
*/
|
|
62
|
+
function isPlumbing(path) {
|
|
63
|
+
return NOT_CONTENT.some((s) => {
|
|
64
|
+
if (s.endsWith('/'))
|
|
65
|
+
return path.includes(s);
|
|
66
|
+
for (let at = path.indexOf(s); at !== -1; at = path.indexOf(s, at + 1)) {
|
|
67
|
+
const next = path.charAt(at + s.length);
|
|
68
|
+
if (next === '' || next === '/' || next === '.')
|
|
69
|
+
return true;
|
|
70
|
+
}
|
|
71
|
+
return false;
|
|
72
|
+
});
|
|
73
|
+
}
|
|
74
|
+
/**
|
|
75
|
+
* One canonical spelling of a URL, or null if it is not a page address at all.
|
|
76
|
+
*
|
|
77
|
+
* The trailing slash, `index.html` and a fragment are the three ways the same
|
|
78
|
+
* page arrives under different names, and a crawl that does not fold them
|
|
79
|
+
* imports the home page four times. The query string is DELIBERATELY kept here
|
|
80
|
+
* and dropped later, so the drop can be counted and reported rather than
|
|
81
|
+
* happening invisibly.
|
|
82
|
+
*/
|
|
83
|
+
export function normalizeUrl(raw, base) {
|
|
84
|
+
let u;
|
|
85
|
+
try {
|
|
86
|
+
u = base ? new URL(raw, base) : new URL(raw);
|
|
87
|
+
}
|
|
88
|
+
catch {
|
|
89
|
+
// A NON-HIERARCHICAL BASE THROWS, and the throw is the whole reason this is
|
|
90
|
+
// wrapped: `new URL('/a', 'data:text/html,…')` is a TypeError, and one bad
|
|
91
|
+
// href on one page must not end a crawl.
|
|
92
|
+
return null;
|
|
93
|
+
}
|
|
94
|
+
if (u.protocol !== 'http:' && u.protocol !== 'https:')
|
|
95
|
+
return null;
|
|
96
|
+
u.hash = '';
|
|
97
|
+
u.username = '';
|
|
98
|
+
u.password = '';
|
|
99
|
+
let path = u.pathname.replace(/\/index\.[a-z]{2,5}$/i, '/');
|
|
100
|
+
if (path.length > 1)
|
|
101
|
+
path = path.replace(/\/+$/, '');
|
|
102
|
+
u.pathname = path === '' ? '/' : path;
|
|
103
|
+
return u.toString();
|
|
104
|
+
}
|
|
105
|
+
function originOf(url) {
|
|
106
|
+
try {
|
|
107
|
+
return new URL(url).origin;
|
|
108
|
+
}
|
|
109
|
+
catch {
|
|
110
|
+
return '';
|
|
111
|
+
}
|
|
112
|
+
}
|
|
113
|
+
/**
|
|
114
|
+
* The path, as a human wrote it.
|
|
115
|
+
*
|
|
116
|
+
* `URL` percent-encodes `pathname`, so a Vietnamese path comes back as
|
|
117
|
+
* `/trang-ch%E1%BB%A7` — which turns into `trang-ch-e1-bb-a7` the moment a slug
|
|
118
|
+
* is derived from it, and makes an `include: ['/tin-tức']` match nothing. The
|
|
119
|
+
* decode is guarded because a malformed sequence throws, and a path this cannot
|
|
120
|
+
* read is better matched raw than not at all.
|
|
121
|
+
*/
|
|
122
|
+
function pathOf(url) {
|
|
123
|
+
let raw;
|
|
124
|
+
try {
|
|
125
|
+
raw = new URL(url).pathname;
|
|
126
|
+
}
|
|
127
|
+
catch {
|
|
128
|
+
return '/';
|
|
129
|
+
}
|
|
130
|
+
try {
|
|
131
|
+
return decodeURIComponent(raw);
|
|
132
|
+
}
|
|
133
|
+
catch {
|
|
134
|
+
return raw;
|
|
135
|
+
}
|
|
136
|
+
}
|
|
137
|
+
/**
|
|
138
|
+
* The `<loc>` values in a sitemap, and the child sitemaps of an index.
|
|
139
|
+
*
|
|
140
|
+
* Read with a regex rather than an XML parser on purpose: the whole grammar
|
|
141
|
+
* this needs is one element name, the file arrives from a stranger's server,
|
|
142
|
+
* and a dependency that parses arbitrary XML from an untrusted origin is a
|
|
143
|
+
* larger surface than the feature is worth. `<sitemapindex>` is the only
|
|
144
|
+
* distinction that matters — its locs are more sitemaps, not pages.
|
|
145
|
+
*/
|
|
146
|
+
export function sitemapUrls(xml) {
|
|
147
|
+
const locs = [];
|
|
148
|
+
for (const m of xml.matchAll(/<loc>\s*([^<]+?)\s*<\/loc>/gi)) {
|
|
149
|
+
const v = m[1]
|
|
150
|
+
.replace(/&/g, '&')
|
|
151
|
+
.replace(/</g, '<')
|
|
152
|
+
.replace(/>/g, '>')
|
|
153
|
+
.replace(/"/g, '"')
|
|
154
|
+
.replace(/'/g, "'");
|
|
155
|
+
locs.push(v);
|
|
156
|
+
}
|
|
157
|
+
const isIndex = /<sitemapindex[\s>]/i.test(xml);
|
|
158
|
+
return isIndex ? { pages: [], sitemaps: locs } : { pages: locs, sitemaps: [] };
|
|
159
|
+
}
|
|
160
|
+
/** The `Sitemap:` lines of a robots.txt — where a site says its sitemap really lives. */
|
|
161
|
+
export function robotsSitemaps(txt) {
|
|
162
|
+
const out = [];
|
|
163
|
+
for (const line of txt.split(/\r?\n/)) {
|
|
164
|
+
const m = /^\s*sitemap\s*:\s*(\S+)/i.exec(line);
|
|
165
|
+
if (m)
|
|
166
|
+
out.push(m[1]);
|
|
167
|
+
}
|
|
168
|
+
return out;
|
|
169
|
+
}
|
|
170
|
+
/** A slug this platform will accept, from a path. The root is the home page. */
|
|
171
|
+
export function slugFor(url, taken) {
|
|
172
|
+
const path = pathOf(url);
|
|
173
|
+
const base = path === '/'
|
|
174
|
+
? 'home'
|
|
175
|
+
: path
|
|
176
|
+
.replace(/\.[a-z0-9]{1,5}$/i, '')
|
|
177
|
+
.split('/')
|
|
178
|
+
.filter(Boolean)
|
|
179
|
+
.join('-');
|
|
180
|
+
let slug = base
|
|
181
|
+
.toLowerCase()
|
|
182
|
+
// NFD does not decompose đ — it is a letter of its own, not a d with a
|
|
183
|
+
// stroke — and this platform is Vietnamese first, so a path with one in it
|
|
184
|
+
// would lose the character rather than transliterate it.
|
|
185
|
+
.replace(/đ/g, 'd')
|
|
186
|
+
.normalize('NFD')
|
|
187
|
+
.replace(/[\u0300-\u036f]/g, '')
|
|
188
|
+
.replace(/[^a-z0-9]+/g, '-')
|
|
189
|
+
.replace(/^-+|-+$/g, '')
|
|
190
|
+
.slice(0, 60)
|
|
191
|
+
.replace(/-+$/g, '');
|
|
192
|
+
if (!slug)
|
|
193
|
+
slug = 'page';
|
|
194
|
+
// A COLLIDING SLUG IS RENAMED BY THE PLATFORM, NOT REFUSED (`uniqueSlug`
|
|
195
|
+
// suffixes -1, -2 … and its own comment says it never errors), so a duplicate
|
|
196
|
+
// inside one run would come back under a name the caller never asked for and
|
|
197
|
+
// every link authored to the requested one would be dead. Settle it here,
|
|
198
|
+
// where the caller can still see both names in the plan.
|
|
199
|
+
let out = slug;
|
|
200
|
+
for (let n = 2; taken.has(out); n += 1)
|
|
201
|
+
out = `${slug.slice(0, 57)}-${n}`;
|
|
202
|
+
taken.add(out);
|
|
203
|
+
return out;
|
|
204
|
+
}
|
|
205
|
+
/** A human name from a slug, for a page whose own `<title>` is not known yet. */
|
|
206
|
+
export function nameFor(slug) {
|
|
207
|
+
if (slug === 'home')
|
|
208
|
+
return 'Home';
|
|
209
|
+
return slug
|
|
210
|
+
.split('-')
|
|
211
|
+
.filter(Boolean)
|
|
212
|
+
.map((w) => w.charAt(0).toUpperCase() + w.slice(1))
|
|
213
|
+
.join(' ');
|
|
214
|
+
}
|
|
215
|
+
/**
|
|
216
|
+
* Decide which of the discovered URLs become pages, and in what order.
|
|
217
|
+
*
|
|
218
|
+
* ORDER IS PART OF THE ANSWER, not a detail. A sitemap can list five thousand
|
|
219
|
+
* URLs and the cap will take a dozen; taking the first dozen in file order gives
|
|
220
|
+
* a site made of whatever the generator happened to emit first, which on a shop
|
|
221
|
+
* is twelve product pages and no home page. Shallowest first — the root, then
|
|
222
|
+
* `/about`, then `/blog/a-post` — is the site's own outline, so the cap keeps
|
|
223
|
+
* the pages a visitor would actually be shown.
|
|
224
|
+
*/
|
|
225
|
+
export function choosePages(entry, urls, opts = {}) {
|
|
226
|
+
const maxPages = opts.maxPages ?? 12;
|
|
227
|
+
const include = (opts.include ?? []).map((s) => s.toLowerCase());
|
|
228
|
+
const exclude = (opts.exclude ?? []).map((s) => s.toLowerCase());
|
|
229
|
+
const origin = originOf(entry);
|
|
230
|
+
const skipped = {};
|
|
231
|
+
const skip = (why) => {
|
|
232
|
+
skipped[why] = (skipped[why] ?? 0) + 1;
|
|
233
|
+
};
|
|
234
|
+
const seen = new Set();
|
|
235
|
+
const kept = [];
|
|
236
|
+
// THE ENTRY IS PAGE ONE, whether or not the discovery handed it back. A
|
|
237
|
+
// sitemap that omits the home page is ordinary — plenty of generators list
|
|
238
|
+
// only what they manage — and a caller who pasted a link and got a site
|
|
239
|
+
// without the page they pasted has been given the wrong site. It still passes
|
|
240
|
+
// through `include` / `exclude` below: forcing it past the caller's own filter
|
|
241
|
+
// would be a different kind of wrong.
|
|
242
|
+
for (const f of [{ url: entry, from: 'entry' }, ...urls]) {
|
|
243
|
+
const norm = normalizeUrl(f.url);
|
|
244
|
+
if (!norm) {
|
|
245
|
+
skip('unreadable');
|
|
246
|
+
continue;
|
|
247
|
+
}
|
|
248
|
+
if (originOf(norm) !== origin) {
|
|
249
|
+
skip('off-site');
|
|
250
|
+
continue;
|
|
251
|
+
}
|
|
252
|
+
const path = pathOf(norm).toLowerCase();
|
|
253
|
+
const wanted = include.length > 0 ? include.some((s) => path.includes(s)) : null;
|
|
254
|
+
if (wanted === false) {
|
|
255
|
+
skip('not-included');
|
|
256
|
+
continue;
|
|
257
|
+
}
|
|
258
|
+
if (exclude.some((s) => path.includes(s))) {
|
|
259
|
+
skip('excluded');
|
|
260
|
+
continue;
|
|
261
|
+
}
|
|
262
|
+
// AN EXPLICIT include OUTRANKS THE PLUMBING LIST. The list is a heuristic
|
|
263
|
+
// about a stranger's site, and a caller who names `/account` knows something
|
|
264
|
+
// this module does not.
|
|
265
|
+
if (!wanted && norm !== entry) {
|
|
266
|
+
if (ASSET.test(path)) {
|
|
267
|
+
skip('asset');
|
|
268
|
+
continue;
|
|
269
|
+
}
|
|
270
|
+
if (isPlumbing(path)) {
|
|
271
|
+
skip('not-content');
|
|
272
|
+
continue;
|
|
273
|
+
}
|
|
274
|
+
}
|
|
275
|
+
if (new URL(norm).search) {
|
|
276
|
+
// A query string is nearly always a filter, a sort or a page number over
|
|
277
|
+
// content already taken, and following them is how a crawl of a shop
|
|
278
|
+
// spends twelve pages on the same grid.
|
|
279
|
+
skip('query');
|
|
280
|
+
continue;
|
|
281
|
+
}
|
|
282
|
+
if (seen.has(norm)) {
|
|
283
|
+
// A SECOND COPY OF THE ENTRY IS NOT A LOSS. It is injected above and the
|
|
284
|
+
// discovery hands it back too — a sitemap lists it, a crawl seeds its
|
|
285
|
+
// frontier with it — so counting that collision reports a page dropped
|
|
286
|
+
// when none was.
|
|
287
|
+
if (norm !== entry)
|
|
288
|
+
skip('duplicate');
|
|
289
|
+
continue;
|
|
290
|
+
}
|
|
291
|
+
seen.add(norm);
|
|
292
|
+
kept.push({ url: norm, from: norm === entry ? 'entry' : f.from });
|
|
293
|
+
}
|
|
294
|
+
// GROUPS ARE COUNTED BEFORE THE CAP, because the whole point of reporting them
|
|
295
|
+
// is to say what the cap is about to hide.
|
|
296
|
+
const groups = {};
|
|
297
|
+
for (const f of kept) {
|
|
298
|
+
const seg = pathOf(f.url).split('/').filter(Boolean)[0];
|
|
299
|
+
if (seg)
|
|
300
|
+
groups[seg] = (groups[seg] ?? 0) + 1;
|
|
301
|
+
}
|
|
302
|
+
for (const k of Object.keys(groups))
|
|
303
|
+
if (groups[k] < 3)
|
|
304
|
+
delete groups[k];
|
|
305
|
+
const depthOf = (u) => pathOf(u).split('/').filter(Boolean).length;
|
|
306
|
+
kept.sort((a, b) => {
|
|
307
|
+
if (a.url === entry)
|
|
308
|
+
return -1;
|
|
309
|
+
if (b.url === entry)
|
|
310
|
+
return 1;
|
|
311
|
+
const d = depthOf(a.url) - depthOf(b.url);
|
|
312
|
+
return d !== 0 ? d : a.url.localeCompare(b.url);
|
|
313
|
+
});
|
|
314
|
+
const over = Math.max(0, kept.length - maxPages);
|
|
315
|
+
if (over > 0)
|
|
316
|
+
skipped['over-page-limit'] = over;
|
|
317
|
+
const taken = new Set();
|
|
318
|
+
const pages = kept.slice(0, maxPages).map((f) => {
|
|
319
|
+
const slug = slugFor(f.url, taken);
|
|
320
|
+
return { ...f, slug, name: nameFor(slug), depth: depthOf(f.url) };
|
|
321
|
+
});
|
|
322
|
+
return { pages, skipped, groups };
|
|
323
|
+
}
|
|
324
|
+
/**
|
|
325
|
+
* The filter a link crawl needs, as one function: a raw href in, its canonical
|
|
326
|
+
* spelling out, or null.
|
|
327
|
+
*
|
|
328
|
+
* Shares the tables with `choosePages` on purpose — a crawl that queued the
|
|
329
|
+
* links `choosePages` is about to throw away would spend its whole navigation
|
|
330
|
+
* budget on a login page and a PDF. `include` / `exclude` are NOT applied here:
|
|
331
|
+
* those are the caller's narrowing of the site, and a page excluded from the
|
|
332
|
+
* import can still be the page that links to one that is not.
|
|
333
|
+
*/
|
|
334
|
+
export function canonFor(entry) {
|
|
335
|
+
const origin = originOf(entry);
|
|
336
|
+
return (raw) => {
|
|
337
|
+
const norm = normalizeUrl(raw, entry);
|
|
338
|
+
if (!norm || originOf(norm) !== origin)
|
|
339
|
+
return null;
|
|
340
|
+
const path = pathOf(norm).toLowerCase();
|
|
341
|
+
if (ASSET.test(path))
|
|
342
|
+
return null;
|
|
343
|
+
if (isPlumbing(path))
|
|
344
|
+
return null;
|
|
345
|
+
return norm;
|
|
346
|
+
};
|
|
347
|
+
}
|
|
@@ -56,6 +56,33 @@ export function checkBandOrder(doc) {
|
|
|
56
56
|
}
|
|
57
57
|
return null;
|
|
58
58
|
}
|
|
59
|
+
/**
|
|
60
|
+
* WHERE NEW PAGE CONTENT MAY BE ADDED among ROOT's children: before the first
|
|
61
|
+
* global footer, or at the end when there is none.
|
|
62
|
+
*
|
|
63
|
+
* APPENDING TO ROOT IS THE OBVIOUS THING AND IT IS WRONG ON A REAL PAGE. Every
|
|
64
|
+
* site that has a global footer has one as a ROOT child, so `data.nodes.length`
|
|
65
|
+
* puts the new section AFTER it — `checkBandOrder` then refuses the save, the
|
|
66
|
+
* platform would have refused it too, and the caller is told about a band rule
|
|
67
|
+
* they did not knowingly break. Measured on the one path where it is the
|
|
68
|
+
* DEFAULT: `sb_import_site` imports the entry URL into the site's existing home
|
|
69
|
+
* page, which is exactly the page most likely to carry both globals.
|
|
70
|
+
*
|
|
71
|
+
* The index is into `data.nodes` RAW, overlays included, because that is what
|
|
72
|
+
* `addSubtree` takes. Overlays are skipped when deciding, never when counting:
|
|
73
|
+
* the platform strips them before it checks the bands, so where one sits says
|
|
74
|
+
* nothing about where content may go.
|
|
75
|
+
*/
|
|
76
|
+
export function middleEnd(doc) {
|
|
77
|
+
const kids = doc.nodes[doc.root_node_id]?.data?.nodes ?? [];
|
|
78
|
+
for (let i = 0; i < kids.length; i += 1) {
|
|
79
|
+
if (isOverlay(doc, kids[i]))
|
|
80
|
+
continue;
|
|
81
|
+
if (bandOf(doc, kids[i]) === 'footer')
|
|
82
|
+
return i;
|
|
83
|
+
}
|
|
84
|
+
return kids.length;
|
|
85
|
+
}
|
|
59
86
|
/** Is this node a composed GLOBAL SECTION master — a shared header or footer? */
|
|
60
87
|
export function isGlobal(doc, id) {
|
|
61
88
|
return doc.nodes[id]?.specials?.[SPEC_GLOBAL_ID] !== undefined;
|
package/dist/tools/importpage.js
CHANGED
|
@@ -1,30 +1,152 @@
|
|
|
1
1
|
import { z } from 'zod';
|
|
2
2
|
import { text } from '../mcp/response.js';
|
|
3
|
-
import { capture } from '../vision/capture.js';
|
|
3
|
+
import { capture, captureMany, crawlLinks } from '../vision/capture.js';
|
|
4
4
|
import { uploadMedia } from '../transport/media.js';
|
|
5
5
|
import { addSubtree } from '../domains/site/builder.js';
|
|
6
|
+
import { middleEnd } from '../domains/site/traps.js';
|
|
6
7
|
import { toSpecs, tokensFromPage, imageSources, rehostImages, } from '../domains/site/importmap.js';
|
|
8
|
+
import { canonFor, choosePages, normalizeUrl, robotsSitemaps, sitemapUrls, } from '../domains/site/discover.js';
|
|
9
|
+
import { loadSource } from '../transport/pages.js';
|
|
10
|
+
import { PageDoc } from '../domains/site/document.js';
|
|
11
|
+
import { request } from '../transport/http.js';
|
|
12
|
+
import { siteToken } from './credentialpick.js';
|
|
7
13
|
import { siteFor } from './context.js';
|
|
8
14
|
/**
|
|
9
|
-
*
|
|
15
|
+
* Read a URL from SOMEBODY ELSE'S ORIGIN.
|
|
10
16
|
*
|
|
11
|
-
*
|
|
12
|
-
*
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
-
*
|
|
17
|
+
* NOT through `request()`, and that is the point: every path in that module
|
|
18
|
+
* attaches a credential, and this one must attach none. A sitemap fetch that
|
|
19
|
+
* carried `SB_TOKEN` would hand this install's key to a stranger's server
|
|
20
|
+
* because the caller pasted a link — the same envelope rule this repo keeps for
|
|
21
|
+
* every other secret, applied to the one call that leaves the platform.
|
|
16
22
|
*
|
|
17
|
-
*
|
|
18
|
-
*
|
|
19
|
-
*
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
+
* A failure is an ANSWER, not an error: most sites have no robots.txt, plenty
|
|
24
|
+
* have no sitemap, and discovery falls through to the link crawl. Nothing here
|
|
25
|
+
* is worth ending a tool call over.
|
|
26
|
+
*/
|
|
27
|
+
async function fetchForeign(ctx, url) {
|
|
28
|
+
const f = ctx.fetchImpl ?? fetch;
|
|
29
|
+
try {
|
|
30
|
+
const res = await f(url, {
|
|
31
|
+
signal: AbortSignal.timeout(8_000),
|
|
32
|
+
redirect: 'follow',
|
|
33
|
+
});
|
|
34
|
+
if (!res.ok)
|
|
35
|
+
return null;
|
|
36
|
+
const body = await res.text();
|
|
37
|
+
// A sitemap is text from a stranger. Bounded so a multi-megabyte one costs
|
|
38
|
+
// a slice rather than the process.
|
|
39
|
+
return body.length > 5_000_000 ? body.slice(0, 5_000_000) : body;
|
|
40
|
+
}
|
|
41
|
+
catch {
|
|
42
|
+
return null;
|
|
43
|
+
}
|
|
44
|
+
}
|
|
45
|
+
/**
|
|
46
|
+
* THE PUBLISHER'S OWN LIST OF ITS PAGES, or null if it does not offer one.
|
|
47
|
+
*
|
|
48
|
+
* One fetch and no browser, and it lists pages nothing links to — which is why
|
|
49
|
+
* it is tried before the crawl rather than after it.
|
|
50
|
+
*/
|
|
51
|
+
export async function fromSitemap(ctx, entry) {
|
|
52
|
+
const origin = new URL(entry).origin;
|
|
53
|
+
const declared = new Set();
|
|
54
|
+
// ROBOTS.TXT IS WHERE A SITE SAYS WHERE ITS SITEMAP REALLY IS, and plenty of
|
|
55
|
+
// real ones are not at /sitemap.xml — a shop platform names
|
|
56
|
+
// /sitemap_products_1.xml, a CMS a dated path. Guessing only the default is how
|
|
57
|
+
// a site with a perfectly good sitemap gets crawled instead.
|
|
58
|
+
const robots = await fetchForeign(ctx, `${origin}/robots.txt`);
|
|
59
|
+
if (robots)
|
|
60
|
+
for (const u of robotsSitemaps(robots))
|
|
61
|
+
declared.add(u);
|
|
62
|
+
for (const guess of ['/sitemap.xml', '/sitemap_index.xml', '/sitemap-index.xml']) {
|
|
63
|
+
declared.add(`${origin}${guess}`);
|
|
64
|
+
}
|
|
65
|
+
const pages = new Set();
|
|
66
|
+
let queue = [...declared].slice(0, 5);
|
|
67
|
+
// ONE level of index expansion. A sitemap index of indexes exists and is rare;
|
|
68
|
+
// the bound is what keeps a pathological one from becoming a fetch storm on
|
|
69
|
+
// somebody else's server.
|
|
70
|
+
for (let round = 0; round < 2 && queue.length > 0; round += 1) {
|
|
71
|
+
const next = [];
|
|
72
|
+
for (const sm of queue.slice(0, 5)) {
|
|
73
|
+
const xml = await fetchForeign(ctx, sm);
|
|
74
|
+
if (!xml)
|
|
75
|
+
continue;
|
|
76
|
+
const got = sitemapUrls(xml);
|
|
77
|
+
for (const u of got.pages)
|
|
78
|
+
pages.add(u);
|
|
79
|
+
for (const u of got.sitemaps)
|
|
80
|
+
next.push(u);
|
|
81
|
+
}
|
|
82
|
+
queue = next;
|
|
83
|
+
}
|
|
84
|
+
// TWO IS THE THRESHOLD, NOT ONE. A sitemap listing only the home page is what a
|
|
85
|
+
// half-configured generator emits, and taking it would import a one-page site
|
|
86
|
+
// off a site that has forty.
|
|
87
|
+
if (pages.size < 2)
|
|
88
|
+
return null;
|
|
89
|
+
// THE ENTRY IS NOT PUSHED HERE. `choosePages` guarantees it is page one, so
|
|
90
|
+
// adding it would arrive as a second copy and be counted as a dropped
|
|
91
|
+
// duplicate — a phantom loss on every sitemap run.
|
|
92
|
+
const urls = [];
|
|
93
|
+
for (const u of pages) {
|
|
94
|
+
// NORMALIZED, NOT FILTERED. `canonFor` exists to stop a crawl spending a
|
|
95
|
+
// NAVIGATION on a PDF; a sitemap costs no navigation, so there is nothing to
|
|
96
|
+
// save by dropping one here — and dropping it here means `choosePages` never
|
|
97
|
+
// sees it and never counts the reason. Filtering twice and reporting once is
|
|
98
|
+
// how a caller ends up asking why the cart page vanished.
|
|
99
|
+
const c = normalizeUrl(u, entry);
|
|
100
|
+
if (c)
|
|
101
|
+
urls.push({ url: c, from: 'sitemap' });
|
|
102
|
+
}
|
|
103
|
+
return urls;
|
|
104
|
+
}
|
|
105
|
+
/**
|
|
106
|
+
* THE PAGES THIS SITE ALREADY HAS — read by the preview as well as the run.
|
|
23
107
|
*
|
|
24
|
-
* The
|
|
25
|
-
*
|
|
26
|
-
*
|
|
108
|
+
* The dry run used to make no platform call at all, which made it cheap and
|
|
109
|
+
* made it LIE: it promised twelve pages on a site where four of those slugs
|
|
110
|
+
* were taken (each of which the run then skips, because the platform renames a
|
|
111
|
+
* collision and answers 200) and never said the entry URL was about to be
|
|
112
|
+
* merged into an existing home page rather than given one of its own. A preview
|
|
113
|
+
* whose count does not survive contact with the run is not a preview.
|
|
27
114
|
*/
|
|
115
|
+
async function existingPages(ctx, siteId) {
|
|
116
|
+
const listed = (await request({
|
|
117
|
+
base: ctx.base,
|
|
118
|
+
method: 'GET',
|
|
119
|
+
path: `/api/sites/${encodeURIComponent(siteId)}/pages`,
|
|
120
|
+
token: siteToken(ctx),
|
|
121
|
+
fetchImpl: ctx.fetchImpl,
|
|
122
|
+
}));
|
|
123
|
+
return Array.isArray(listed.pages) ? listed.pages : [];
|
|
124
|
+
}
|
|
125
|
+
/**
|
|
126
|
+
* WHAT PAGES THIS SITE HAS — the publisher's own answer first, a crawl second.
|
|
127
|
+
*
|
|
128
|
+
* The order is about cost, not preference. A sitemap is one fetch and no browser;
|
|
129
|
+
* a crawl is a browser navigation per page and can only find what the entry page
|
|
130
|
+
* points at. So the crawl is the fallback, automatically: there is no knob to
|
|
131
|
+
* force it, because a caller who wants fewer pages than the sitemap offers wants
|
|
132
|
+
* `include` or `max_pages`, not a slower way to find the same list.
|
|
133
|
+
*/
|
|
134
|
+
async function discoverSite(ctx, entry, opts) {
|
|
135
|
+
const listed = await fromSitemap(ctx, entry);
|
|
136
|
+
if (listed)
|
|
137
|
+
return { source: 'sitemap', urls: listed, titles: new Map(), visited: 0 };
|
|
138
|
+
const crawled = await crawlLinks(entry, {
|
|
139
|
+
depth: opts.depth ?? 1,
|
|
140
|
+
maxVisits: opts.maxVisits ?? 24,
|
|
141
|
+
canon: canonFor(entry),
|
|
142
|
+
});
|
|
143
|
+
return {
|
|
144
|
+
source: 'links',
|
|
145
|
+
urls: crawled.urls.map((u) => ({ url: u, from: u === entry ? 'entry' : 'links' })),
|
|
146
|
+
titles: crawled.titles,
|
|
147
|
+
visited: crawled.visited,
|
|
148
|
+
};
|
|
149
|
+
}
|
|
28
150
|
export function registerImportTools(server, ctx, session) {
|
|
29
151
|
server.registerTool('sb_import', {
|
|
30
152
|
description: 'Read a page from any public URL and add its structure and content to the OPEN page as ' +
|
|
@@ -133,7 +255,12 @@ export function registerImportTools(server, ctx, session) {
|
|
|
133
255
|
const all = [];
|
|
134
256
|
const staged = doc.preview([]);
|
|
135
257
|
for (const spec of specs) {
|
|
136
|
-
|
|
258
|
+
// BEFORE THE GLOBAL FOOTER, not after it. Appending to ROOT is the
|
|
259
|
+
// obvious thing and it breaks trap 3 on every page that has a footer —
|
|
260
|
+
// the platform refuses the whole save, and the caller is told about a
|
|
261
|
+
// band rule they did not knowingly break. Recomputed each time because
|
|
262
|
+
// the last insert moved it.
|
|
263
|
+
const { patches, ids } = addSubtree(staged, staged.doc.root_node_id, spec, middleEnd(staged.doc));
|
|
137
264
|
staged.apply(patches);
|
|
138
265
|
all.push(...patches);
|
|
139
266
|
added.push(ids[0]);
|
|
@@ -160,4 +287,289 @@ export function registerImportTools(server, ctx, session) {
|
|
|
160
287
|
: ''),
|
|
161
288
|
});
|
|
162
289
|
});
|
|
290
|
+
server.registerTool('sb_import_site', {
|
|
291
|
+
description: 'Read a WHOLE site from one URL — its sitemap, or the links on that page — and give each ' +
|
|
292
|
+
'page found its own DRAFT page here, built from this site\'s tokens. Not a clone. Dry run ' +
|
|
293
|
+
'returns the page list before anything is created.',
|
|
294
|
+
inputSchema: {
|
|
295
|
+
url: z.string().describe('Any page of the site'),
|
|
296
|
+
site_id: z.string().optional(),
|
|
297
|
+
max_pages: z.number().int().min(1).max(60).optional().describe('Default 12'),
|
|
298
|
+
depth: z.number().int().min(0).max(3).optional().describe('No sitemap: link depth, default 1'),
|
|
299
|
+
include: z.array(z.string()).optional().describe('Path substrings to keep'),
|
|
300
|
+
exclude: z.array(z.string()).optional(),
|
|
301
|
+
max_images: z.number().int().min(0).max(200).optional().describe('Default 24, whole import'),
|
|
302
|
+
max_nodes: z.number().int().min(1).max(1000).optional().describe('Per page, default 300'),
|
|
303
|
+
upload_images: z.boolean().optional(),
|
|
304
|
+
homepage: z.boolean().optional().describe("Entry into this site's home page, default true"),
|
|
305
|
+
dry_run: z.boolean().optional(),
|
|
306
|
+
},
|
|
307
|
+
annotations: { readOnlyHint: false, destructiveHint: false, openWorldHint: true },
|
|
308
|
+
}, async ({ url, site_id: given, max_pages, depth, include, exclude, max_images, max_nodes, upload_images, homepage, dry_run, }) => {
|
|
309
|
+
const siteId = siteFor(ctx, given);
|
|
310
|
+
const entry = normalizeUrl(url);
|
|
311
|
+
if (!entry) {
|
|
312
|
+
throw new Error(`sbuilder: "${url}" is not a page address this server can read — an http or https URL is needed.`);
|
|
313
|
+
}
|
|
314
|
+
const found = await discoverSite(ctx, entry, {
|
|
315
|
+
depth,
|
|
316
|
+
// The crawl may look at more pages than it imports — that is how it finds
|
|
317
|
+
// the twelfth — but not without bound.
|
|
318
|
+
maxVisits: Math.max(4, (max_pages ?? 12) * 2),
|
|
319
|
+
});
|
|
320
|
+
const chosen = choosePages(entry, found.urls, {
|
|
321
|
+
maxPages: max_pages,
|
|
322
|
+
include,
|
|
323
|
+
exclude,
|
|
324
|
+
});
|
|
325
|
+
// A CRAWL ALREADY READ THE TITLE. `nameFor` derives a name from the slug
|
|
326
|
+
// because a sitemap offers nothing else, but the link crawl opened every
|
|
327
|
+
// one of these pages to read its links and has the page's own `<title>` —
|
|
328
|
+
// so the plan can name them the way their author does, before anything is
|
|
329
|
+
// captured.
|
|
330
|
+
const plan = {
|
|
331
|
+
...chosen,
|
|
332
|
+
pages: chosen.pages.map((p) => ({ ...p, name: found.titles.get(p.url) || p.name })),
|
|
333
|
+
};
|
|
334
|
+
if (plan.pages.length === 0) {
|
|
335
|
+
throw new Error(`sbuilder: no page worth importing was found from ${entry}. Discovered by ` +
|
|
336
|
+
`${found.source}; skipped ${JSON.stringify(plan.skipped)}. A site behind a login, or ` +
|
|
337
|
+
'one whose links are all off-site, reads as empty here.');
|
|
338
|
+
}
|
|
339
|
+
// FORTY URLS UNDER ONE PREFIX ARE NOT FORTY PAGES ON THIS PLATFORM.
|
|
340
|
+
//
|
|
341
|
+
// They are one entity TEMPLATE plus a catalogue: `/products/x` resolves to
|
|
342
|
+
// the site's published page of type `product`, bound to the record in the
|
|
343
|
+
// URL. Importing them as static pages produces a shop where every price is
|
|
344
|
+
// a literal, nothing is buyable, and `sb_review` reports a missing purchase
|
|
345
|
+
// action on forty pages at once. Said before anything is created, because
|
|
346
|
+
// after it the fix is forty deletes.
|
|
347
|
+
const heavy = Object.entries(plan.groups)
|
|
348
|
+
.filter(([, n]) => n >= 3)
|
|
349
|
+
.map(([seg, n]) => `${seg} (${n})`);
|
|
350
|
+
const templateNote = heavy.length > 0
|
|
351
|
+
? `Several URLs share a prefix — ${heavy.join(', ')}. If those are products, ` +
|
|
352
|
+
'collections or posts, they are ONE template plus real records here, not one page ' +
|
|
353
|
+
'each: sb_page_create type:"product" (or category/post) seeds the bound page, and ' +
|
|
354
|
+
'the catalogue comes from the API. Pass exclude to leave them out of the import.'
|
|
355
|
+
: undefined;
|
|
356
|
+
// BEST EFFORT IN THE PREVIEW, REQUIRED IN THE RUN.
|
|
357
|
+
//
|
|
358
|
+
// The listing is what makes the preview honest — which slugs are taken,
|
|
359
|
+
// whether the entry merges into an existing home page — but demanding it
|
|
360
|
+
// would turn "what is on that website?" into a question only a connected
|
|
361
|
+
// install may ask, and the discovery half needs no credential at all. So a
|
|
362
|
+
// dry run that cannot read the site says the landing spot is unknown
|
|
363
|
+
// rather than inventing one; the real run must not guess, and rethrows.
|
|
364
|
+
let existing = [];
|
|
365
|
+
let unlistable = '';
|
|
366
|
+
try {
|
|
367
|
+
existing = await existingPages(ctx, siteId);
|
|
368
|
+
}
|
|
369
|
+
catch (e) {
|
|
370
|
+
// NAME WHAT FAILED. The bare platform string ("boom", "not found") tells
|
|
371
|
+
// a caller who asked to import a website nothing about WHICH call broke,
|
|
372
|
+
// and the one that broke is this site's own page listing — without it the
|
|
373
|
+
// run cannot tell a free slug from a taken one, or find the home page.
|
|
374
|
+
if (dry_run === false) {
|
|
375
|
+
throw new Error(`sbuilder: could not read this site's own pages, so the import cannot tell which ` +
|
|
376
|
+
`slugs are free or which page is the home page — nothing was created. ` +
|
|
377
|
+
`${e.message.replace(/^sbuilder:\s*/, '')}`);
|
|
378
|
+
}
|
|
379
|
+
unlistable = e.message.replace(/^sbuilder:\s*/, '');
|
|
380
|
+
}
|
|
381
|
+
const home = existing.find((e) => e.isHomepage === true);
|
|
382
|
+
const taken = new Set(existing.map((e) => (typeof e.slug === 'string' ? e.slug : '')).filter(Boolean));
|
|
383
|
+
const lands = (p) => unlistable
|
|
384
|
+
? {}
|
|
385
|
+
: p.url === entry && homepage !== false && home
|
|
386
|
+
? { into: 'the existing home page' }
|
|
387
|
+
: taken.has(p.slug)
|
|
388
|
+
? { conflict: `a page with slug "${p.slug}" already exists — this one is SKIPPED` }
|
|
389
|
+
: {};
|
|
390
|
+
if (dry_run !== false) {
|
|
391
|
+
return text({
|
|
392
|
+
dry_run: true,
|
|
393
|
+
entry,
|
|
394
|
+
discovered_by: found.source,
|
|
395
|
+
...(found.visited ? { pages_read_to_find_them: found.visited } : {}),
|
|
396
|
+
pages: plan.pages.map((p) => ({ url: p.url, slug: p.slug, name: p.name, ...lands(p) })),
|
|
397
|
+
...(Object.keys(plan.skipped).length ? { skipped: plan.skipped } : {}),
|
|
398
|
+
...(unlistable
|
|
399
|
+
? {
|
|
400
|
+
landing_unknown: `This site's own pages could not be read (${unlistable}), so which of the above ` +
|
|
401
|
+
'would merge into an existing home page, and which would collide with a slug ' +
|
|
402
|
+
'already taken, is not known yet.',
|
|
403
|
+
}
|
|
404
|
+
: {}),
|
|
405
|
+
...(templateNote ? { entity_pages: templateNote } : {}),
|
|
406
|
+
note: 'Nothing has been created. Each page above becomes a DRAFT page here, filled with the ' +
|
|
407
|
+
"source's structure and text and styled with this site's own tokens — the source's CSS " +
|
|
408
|
+
'and layout are not copied. Pass dry_run:false to build them.',
|
|
409
|
+
});
|
|
410
|
+
}
|
|
411
|
+
// THE TOKENS COME FROM THIS SITE, ONCE, FOR EVERY IMPORTED PAGE.
|
|
412
|
+
//
|
|
413
|
+
// `sb_import` reads them off the OPEN page, which is right when the import
|
|
414
|
+
// is one section onto a page that already has a look. Here most of the
|
|
415
|
+
// target pages do not exist yet and the ones that do are blank, so reading
|
|
416
|
+
// per page would give the first page element defaults and every later page
|
|
417
|
+
// the defaults of the blank page before it — rule 0 failing on every page
|
|
418
|
+
// at once. The open page if there is one, the site's home page otherwise.
|
|
419
|
+
let tokenDoc = session.peek();
|
|
420
|
+
if (!tokenDoc && home && typeof home.id === 'string') {
|
|
421
|
+
try {
|
|
422
|
+
tokenDoc = PageDoc.from((await loadSource(ctx, siteId, home.id)).document);
|
|
423
|
+
}
|
|
424
|
+
catch {
|
|
425
|
+
// An unreadable home page costs the tokens, not the import: every field
|
|
426
|
+
// of PageTokens is optional and falls back to the element's own
|
|
427
|
+
// defaults, which is the same answer an empty target gives.
|
|
428
|
+
}
|
|
429
|
+
}
|
|
430
|
+
const tokens = tokenDoc ? tokensFromPage(tokenDoc.doc) : {};
|
|
431
|
+
const shots = await captureMany(plan.pages.map((p) => p.url), { maxImages: max_images ?? 24, maxNodes: max_nodes ?? 300 });
|
|
432
|
+
const byUrl = new Map(shots.map((s) => [s.url, s]));
|
|
433
|
+
// ONE UPLOAD PER IMAGE FOR THE WHOLE SITE, not per page. A logo, a payment
|
|
434
|
+
// strip and a footer badge appear on every page of a real site, and
|
|
435
|
+
// uploading each of them twelve times would fill the merchant's library
|
|
436
|
+
// with twelve copies and pay twelve round trips for one asset.
|
|
437
|
+
const budget = max_images ?? 24;
|
|
438
|
+
const seenSrc = new Set();
|
|
439
|
+
for (const s of shots) {
|
|
440
|
+
if (!s.ok)
|
|
441
|
+
continue;
|
|
442
|
+
for (const src of imageSources(s.result.sections))
|
|
443
|
+
seenSrc.add(src);
|
|
444
|
+
}
|
|
445
|
+
const wanted = [...seenSrc].slice(0, budget);
|
|
446
|
+
const overBudget = seenSrc.size - wanted.length;
|
|
447
|
+
const rehosted = new Map();
|
|
448
|
+
const failures = new Map();
|
|
449
|
+
if (upload_images !== false) {
|
|
450
|
+
for (const src of wanted) {
|
|
451
|
+
try {
|
|
452
|
+
const up = await uploadMedia(ctx, siteId, { url: src });
|
|
453
|
+
if (up.url)
|
|
454
|
+
rehosted.set(src, up.url);
|
|
455
|
+
}
|
|
456
|
+
catch (e) {
|
|
457
|
+
const why = e.message.replace(/^sbuilder:\s*/, '').slice(0, 160);
|
|
458
|
+
failures.set(why, (failures.get(why) ?? 0) + 1);
|
|
459
|
+
}
|
|
460
|
+
}
|
|
461
|
+
}
|
|
462
|
+
const failedImages = [...failures.entries()].map(([reason, count]) => ({ reason, count }));
|
|
463
|
+
// A PAGE THAT FAILS DOES NOT END THE RUN, and this is the one place in the
|
|
464
|
+
// server where that is the right call. A site import is not atomic and
|
|
465
|
+
// cannot be — each page is its own create and its own save — so the honest
|
|
466
|
+
// shape is per-page outcomes. Aborting on the fourth of twelve would leave
|
|
467
|
+
// three pages built, nine not, and no report saying which.
|
|
468
|
+
const built = [];
|
|
469
|
+
const failed = [];
|
|
470
|
+
let lastOpened = '';
|
|
471
|
+
for (const p of plan.pages) {
|
|
472
|
+
const shot = byUrl.get(p.url);
|
|
473
|
+
if (!shot || !shot.ok) {
|
|
474
|
+
failed.push({ url: p.url, why: shot ? shot.why : 'was not read' });
|
|
475
|
+
continue;
|
|
476
|
+
}
|
|
477
|
+
try {
|
|
478
|
+
const sections = rehosted.size > 0 ? rehostImages(shot.result.sections, rehosted) : shot.result.sections;
|
|
479
|
+
const specs = toSpecs(sections, tokens);
|
|
480
|
+
if (specs.length === 0) {
|
|
481
|
+
failed.push({
|
|
482
|
+
url: p.url,
|
|
483
|
+
why: `nothing renderable — skipped ${JSON.stringify(shot.result.skipped)}`,
|
|
484
|
+
});
|
|
485
|
+
continue;
|
|
486
|
+
}
|
|
487
|
+
let pageId = '';
|
|
488
|
+
let into;
|
|
489
|
+
const isEntry = p.url === entry;
|
|
490
|
+
if (isEntry && homepage !== false && home && typeof home.id === 'string') {
|
|
491
|
+
pageId = home.id;
|
|
492
|
+
into = 'the existing home page';
|
|
493
|
+
}
|
|
494
|
+
else {
|
|
495
|
+
// A COLLIDING SLUG IS RENAMED BY THE PLATFORM, NOT REFUSED, so
|
|
496
|
+
// creating over one answers 200 under a name nobody asked for. On a
|
|
497
|
+
// second run of this tool that would silently double the site.
|
|
498
|
+
if (taken.has(p.slug)) {
|
|
499
|
+
failed.push({
|
|
500
|
+
url: p.url,
|
|
501
|
+
why: `a page with slug "${p.slug}" already exists — left alone, because the ` +
|
|
502
|
+
'platform would have stored this one under a different slug and reported success',
|
|
503
|
+
});
|
|
504
|
+
continue;
|
|
505
|
+
}
|
|
506
|
+
const made = (await request({
|
|
507
|
+
base: ctx.base,
|
|
508
|
+
method: 'POST',
|
|
509
|
+
path: `/api/sites/${encodeURIComponent(siteId)}/pages`,
|
|
510
|
+
token: siteToken(ctx),
|
|
511
|
+
// BLANK, DELIBERATELY. `type: "page"` has no seed, and a seeded
|
|
512
|
+
// page would mix the platform's own content with the imported
|
|
513
|
+
// page's — two headings, two heroes, and no way to tell them apart.
|
|
514
|
+
body: { name: shot.result.title || p.name, type: 'page', slug: p.slug },
|
|
515
|
+
fetchImpl: ctx.fetchImpl,
|
|
516
|
+
}));
|
|
517
|
+
if (typeof made.page?.id !== 'string' || !made.page.id) {
|
|
518
|
+
failed.push({ url: p.url, why: 'the platform created no page for it' });
|
|
519
|
+
continue;
|
|
520
|
+
}
|
|
521
|
+
pageId = made.page.id;
|
|
522
|
+
if (typeof made.page.slug === 'string')
|
|
523
|
+
taken.add(made.page.slug);
|
|
524
|
+
}
|
|
525
|
+
await session.open(siteId, pageId);
|
|
526
|
+
const doc = session.current();
|
|
527
|
+
// STAGED ON A COPY, committed once — the same reason `sb_import` does
|
|
528
|
+
// it: each section's patches are computed from the tree the last one
|
|
529
|
+
// left, and applying them to the real document means a refusal halfway
|
|
530
|
+
// leaves a page half imported with nothing saying which half.
|
|
531
|
+
const staged = doc.preview([]);
|
|
532
|
+
const all = [];
|
|
533
|
+
const added = [];
|
|
534
|
+
for (const spec of specs) {
|
|
535
|
+
const { patches, ids } = addSubtree(staged, staged.doc.root_node_id, spec, middleEnd(staged.doc));
|
|
536
|
+
staged.apply(patches);
|
|
537
|
+
all.push(...patches);
|
|
538
|
+
added.push(ids[0]);
|
|
539
|
+
}
|
|
540
|
+
await session.applyAndSave(all);
|
|
541
|
+
lastOpened = pageId;
|
|
542
|
+
built.push({
|
|
543
|
+
url: p.url,
|
|
544
|
+
slug: p.slug,
|
|
545
|
+
page_id: pageId,
|
|
546
|
+
sections: added.length,
|
|
547
|
+
...(into ? { into } : {}),
|
|
548
|
+
...(Object.keys(shot.result.skipped).length ? { skipped: shot.result.skipped } : {}),
|
|
549
|
+
});
|
|
550
|
+
}
|
|
551
|
+
catch (e) {
|
|
552
|
+
failed.push({ url: p.url, why: e.message.replace(/^sbuilder:\s*/, '').slice(0, 200) });
|
|
553
|
+
}
|
|
554
|
+
}
|
|
555
|
+
return text({
|
|
556
|
+
entry,
|
|
557
|
+
discovered_by: found.source,
|
|
558
|
+
built,
|
|
559
|
+
...(failed.length ? { failed } : {}),
|
|
560
|
+
...(Object.keys(plan.skipped).length ? { skipped: plan.skipped } : {}),
|
|
561
|
+
images: {
|
|
562
|
+
copied: rehosted.size,
|
|
563
|
+
...(failedImages.length ? { failed: failedImages } : {}),
|
|
564
|
+
...(overBudget > 0 ? { over_budget: overBudget } : {}),
|
|
565
|
+
},
|
|
566
|
+
...(lastOpened ? { open: lastOpened } : {}),
|
|
567
|
+
...(templateNote ? { entity_pages: templateNote } : {}),
|
|
568
|
+
directive: ctx.notices.once('import_site', 'These pages are DRAFTS: nothing is live until sb_publish. Three things the import ' +
|
|
569
|
+
"cannot do for you — the source's header and footer were skipped on purpose (this " +
|
|
570
|
+
'site has its own as globals, and a second menu pointing at somebody else\'s site is ' +
|
|
571
|
+
'worse than none), no menu links the new pages together, and nothing has been seen at ' +
|
|
572
|
+
'390px yet. sb_look each page at the three widths before publishing.'),
|
|
573
|
+
});
|
|
574
|
+
});
|
|
163
575
|
}
|
package/dist/tools/page.js
CHANGED
|
@@ -159,6 +159,18 @@ export class PageSession {
|
|
|
159
159
|
throw new Error('sbuilder: no page is open — call sb_page_open first');
|
|
160
160
|
return this.doc;
|
|
161
161
|
}
|
|
162
|
+
/**
|
|
163
|
+
* The open document, or null.
|
|
164
|
+
*
|
|
165
|
+
* `current()` throws, correctly: every editing tool needs a page and the
|
|
166
|
+
* message names the call that opens one. A site import is the one caller for
|
|
167
|
+
* which "no page open" is an ordinary answer rather than a mistake — it reads
|
|
168
|
+
* the design tokens off whatever page is open, and falls back to the site's
|
|
169
|
+
* home page when the caller has not opened one.
|
|
170
|
+
*/
|
|
171
|
+
peek() {
|
|
172
|
+
return this.doc;
|
|
173
|
+
}
|
|
162
174
|
/**
|
|
163
175
|
* Validate, then save.
|
|
164
176
|
*
|
package/dist/vision/capture.js
CHANGED
|
@@ -183,6 +183,25 @@ function capturePage(limits) {
|
|
|
183
183
|
taken.nodes++;
|
|
184
184
|
return [{ kind: 'list', items }];
|
|
185
185
|
}
|
|
186
|
+
// A CODE BLOCK IS ONE THING, and walking into it produces rubble.
|
|
187
|
+
//
|
|
188
|
+
// Every syntax highlighter wraps each token in its own <span>, so the leaf
|
|
189
|
+
// walk took them one at a time: a twenty-line JSON config arrived as forty
|
|
190
|
+
// separate text nodes — `{`, `"mcpServers"`, `: {` — each its own block on
|
|
191
|
+
// its own line. Measured on a real import; it also ate forty of the node
|
|
192
|
+
// budget to say what one node says.
|
|
193
|
+
//
|
|
194
|
+
// Whitespace is collapsed like any other text because this platform has no
|
|
195
|
+
// code element to preserve it in — the six kinds are what can cross — so
|
|
196
|
+
// the honest translation is one paragraph the merchant can then restyle,
|
|
197
|
+
// not a shredded imitation of a listing.
|
|
198
|
+
if (tag === 'PRE' || tag === 'CODE') {
|
|
199
|
+
const text = clean(el.textContent);
|
|
200
|
+
if (!text)
|
|
201
|
+
return [];
|
|
202
|
+
taken.nodes++;
|
|
203
|
+
return [{ kind: 'text', text }];
|
|
204
|
+
}
|
|
186
205
|
if (tag === 'P' || tag === 'BLOCKQUOTE') {
|
|
187
206
|
const text = clean(el.textContent);
|
|
188
207
|
if (!text)
|
|
@@ -197,12 +216,34 @@ function capturePage(limits) {
|
|
|
197
216
|
cs.display === 'inline-flex' || cs.display === 'inline-grid';
|
|
198
217
|
// A ROW is worth keeping; a column is what the page already is, so
|
|
199
218
|
// wrapping one in a group would add a level that renders identically.
|
|
200
|
-
const
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
219
|
+
const grid = cs.display.indexOf('grid') >= 0;
|
|
220
|
+
const row = grid || cs.flexDirection === 'row' || cs.flexDirection === 'row-reverse';
|
|
221
|
+
// A ROW OF 279 IS NOT A ROW.
|
|
222
|
+
//
|
|
223
|
+
// Measured on a real import of a documentation site: its content wrapper
|
|
224
|
+
// is a GRID, every grid was read as a single row, and the whole page came
|
|
225
|
+
// back as one flex row of 279 columns — each column a sliver, every line
|
|
226
|
+
// of prose broken to one word, and 80 nodes hanging past the viewport.
|
|
227
|
+
// The page was structurally perfect and visually destroyed.
|
|
228
|
+
//
|
|
229
|
+
// A design row is a feature trio, a card shelf, a logo wall: a handful of
|
|
230
|
+
// columns, chosen. Past that the container is not arranging things side
|
|
231
|
+
// by side, it is the page's own content column and the browser is
|
|
232
|
+
// wrapping it — so the honest translation is the stack it already reads
|
|
233
|
+
// as. Twelve is above any real row seen here and far below a content
|
|
234
|
+
// grid.
|
|
235
|
+
const ROW_MAX = 12;
|
|
236
|
+
if (lays && row && kids.length >= 2 && kids.length <= ROW_MAX) {
|
|
204
237
|
taken.nodes++;
|
|
205
|
-
return [{
|
|
238
|
+
return [{
|
|
239
|
+
kind: 'group',
|
|
240
|
+
direction: 'row',
|
|
241
|
+
// A GRID ALWAYS WRAPS — that is what a grid IS — and `flexWrap` reads
|
|
242
|
+
// `nowrap` on one because the property does not apply. Carrying that
|
|
243
|
+
// literally gave the columns nowhere to go at any width.
|
|
244
|
+
wrap: grid || cs.flexWrap === 'wrap',
|
|
245
|
+
children: kids,
|
|
246
|
+
}];
|
|
206
247
|
}
|
|
207
248
|
return kids;
|
|
208
249
|
}
|
|
@@ -302,19 +343,20 @@ function capturePage(limits) {
|
|
|
302
343
|
return { url: here, title: clean(document.title), sections, skipped };
|
|
303
344
|
}
|
|
304
345
|
/**
|
|
305
|
-
*
|
|
346
|
+
* ONE BROWSER FOR THE WHOLE CALL.
|
|
306
347
|
*
|
|
307
|
-
*
|
|
308
|
-
*
|
|
309
|
-
*
|
|
348
|
+
* A site import reads a dozen pages, and launching Chrome once per page pays
|
|
349
|
+
* the launch a dozen times for nothing. It is still a launch of its OWN rather
|
|
350
|
+
* than `shoot.ts`'s pooled one, for the reason that module gives: an import is
|
|
351
|
+
* rare, slow and runs untrusted script, and coupling that to the tool a vision
|
|
352
|
+
* loop calls every few hundred milliseconds is how the fast path gets slow.
|
|
353
|
+
*
|
|
354
|
+
* Closed in a `finally`, always. `shoot()` pools for the process lifetime and a
|
|
355
|
+
* caller that forgets `closeBrowser()` never exits — there is no `beforeExit`
|
|
356
|
+
* rescue, because an open browser connection is precisely what stops the event
|
|
357
|
+
* loop draining.
|
|
310
358
|
*/
|
|
311
|
-
|
|
312
|
-
const limits = {
|
|
313
|
-
maxSections: opts.maxSections ?? 24,
|
|
314
|
-
maxImages: opts.maxImages ?? 24,
|
|
315
|
-
maxTextChars: opts.maxTextChars ?? 1200,
|
|
316
|
-
maxNodes: opts.maxNodes ?? 400,
|
|
317
|
-
};
|
|
359
|
+
async function withBrowser(fn) {
|
|
318
360
|
let browser;
|
|
319
361
|
try {
|
|
320
362
|
browser = await chromium.launch({ channel: 'chrome' });
|
|
@@ -323,9 +365,24 @@ export async function capture(url, opts = {}) {
|
|
|
323
365
|
throw new Error('sbuilder: could not start Chrome to read that page. This server uses the SYSTEM ' +
|
|
324
366
|
`browser (playwright-core, channel "chrome") and did not find one: ${e.message}`);
|
|
325
367
|
}
|
|
368
|
+
try {
|
|
369
|
+
return await fn(browser);
|
|
370
|
+
}
|
|
371
|
+
finally {
|
|
372
|
+
await browser.close().catch(() => undefined);
|
|
373
|
+
}
|
|
374
|
+
}
|
|
375
|
+
/**
|
|
376
|
+
* Open one URL in an existing browser, let it settle, and run one evaluate.
|
|
377
|
+
*
|
|
378
|
+
* `load` plus a short settle rather than `networkidle`, for the same reason
|
|
379
|
+
* `shoot.ts` gives: a page with a poller never goes idle, and waiting for that
|
|
380
|
+
* spends the whole budget on a timeout that cannot resolve.
|
|
381
|
+
*/
|
|
382
|
+
async function readPage(browser, url, width, work) {
|
|
326
383
|
let page;
|
|
327
384
|
try {
|
|
328
|
-
page = await browser.newPage({ viewport: { width
|
|
385
|
+
page = await browser.newPage({ viewport: { width, height: 900 } });
|
|
329
386
|
await page.goto(url, { waitUntil: 'load', timeout: 30_000 });
|
|
330
387
|
// THE SAME SETTLE `sb_look` USES, not a flat sleep. A fixed 600ms is wrong
|
|
331
388
|
// at both ends: example.com is finished long before it, and a page that
|
|
@@ -334,10 +391,128 @@ export async function capture(url, opts = {}) {
|
|
|
334
391
|
// question actually being asked (has the page stopped changing) and answers
|
|
335
392
|
// when it becomes true, bounded so a page that never settles is still read.
|
|
336
393
|
await settleDom(page);
|
|
337
|
-
return
|
|
394
|
+
return await work(page);
|
|
338
395
|
}
|
|
339
396
|
finally {
|
|
340
397
|
await page?.close().catch(() => undefined);
|
|
341
|
-
await browser.close().catch(() => undefined);
|
|
342
398
|
}
|
|
343
399
|
}
|
|
400
|
+
function limitsFrom(opts) {
|
|
401
|
+
return {
|
|
402
|
+
maxSections: opts.maxSections ?? 24,
|
|
403
|
+
maxImages: opts.maxImages ?? 24,
|
|
404
|
+
maxTextChars: opts.maxTextChars ?? 1200,
|
|
405
|
+
maxNodes: opts.maxNodes ?? 400,
|
|
406
|
+
};
|
|
407
|
+
}
|
|
408
|
+
/** Open a URL and capture it. */
|
|
409
|
+
export async function capture(url, opts = {}) {
|
|
410
|
+
const limits = limitsFrom(opts);
|
|
411
|
+
return withBrowser((browser) => readPage(browser, url, opts.width ?? 1440, (page) => page.evaluate(capturePage, limits)));
|
|
412
|
+
}
|
|
413
|
+
/**
|
|
414
|
+
* Capture several pages through one browser.
|
|
415
|
+
*
|
|
416
|
+
* A PAGE THAT FAILS MUST NOT END THE RUN. Half the reason to import a site
|
|
417
|
+
* rather than a page is that the caller does not know what is at each URL: one
|
|
418
|
+
* of them is behind a login, one 404s, one hangs. Throwing would discard the
|
|
419
|
+
* eleven that read fine and give the caller nothing to act on, so each outcome
|
|
420
|
+
* is carried and the tool reports the failures by reason.
|
|
421
|
+
*/
|
|
422
|
+
export async function captureMany(urls, opts = {}) {
|
|
423
|
+
const limits = limitsFrom(opts);
|
|
424
|
+
return withBrowser(async (browser) => {
|
|
425
|
+
const out = [];
|
|
426
|
+
for (const url of urls) {
|
|
427
|
+
try {
|
|
428
|
+
const result = (await readPage(browser, url, opts.width ?? 1440, (page) => page.evaluate(capturePage, limits)));
|
|
429
|
+
out.push({ url, ok: true, result });
|
|
430
|
+
}
|
|
431
|
+
catch (e) {
|
|
432
|
+
out.push({ url, ok: false, why: e.message.slice(0, 160) });
|
|
433
|
+
}
|
|
434
|
+
}
|
|
435
|
+
return out;
|
|
436
|
+
});
|
|
437
|
+
}
|
|
438
|
+
/**
|
|
439
|
+
* Every link on a page, absolute.
|
|
440
|
+
*
|
|
441
|
+
* A SECOND, TINY EVALUATE RATHER THAN A FIELD ON `capturePage`. Discovery
|
|
442
|
+
* visits pages the import may never take — that is what a depth-2 crawl IS —
|
|
443
|
+
* and running the whole leaf walk on each of them would pay for content that is
|
|
444
|
+
* thrown away. This asks the one question discovery has.
|
|
445
|
+
*
|
|
446
|
+
* Everything it uses is declared INSIDE it: the function is serialized, so a
|
|
447
|
+
* module-level constant it closes over simply is not there on the other side.
|
|
448
|
+
*/
|
|
449
|
+
function linksOnPage() {
|
|
450
|
+
const here = location.href;
|
|
451
|
+
const links = [];
|
|
452
|
+
const anchors = Array.from(document.querySelectorAll('a[href]'));
|
|
453
|
+
for (const a of anchors) {
|
|
454
|
+
const h = a.getAttribute('href');
|
|
455
|
+
if (!h)
|
|
456
|
+
continue;
|
|
457
|
+
try {
|
|
458
|
+
links.push(new URL(h, here).href);
|
|
459
|
+
}
|
|
460
|
+
catch {
|
|
461
|
+
// `new URL(rel, base)` THROWS on a non-hierarchical base (a data: page).
|
|
462
|
+
// One unresolvable href must not kill the crawl.
|
|
463
|
+
}
|
|
464
|
+
}
|
|
465
|
+
return { title: document.title, links };
|
|
466
|
+
}
|
|
467
|
+
/**
|
|
468
|
+
* Walk a site's own links from one entry page, breadth first.
|
|
469
|
+
*
|
|
470
|
+
* THE FALLBACK, never the first choice — every step is a browser navigation, so
|
|
471
|
+
* the sitemap path exists to avoid this entirely. Bounded on both axes: `depth`
|
|
472
|
+
* limits how far from the entry a page may be, `maxVisits` limits how many
|
|
473
|
+
* navigations the whole crawl may spend, and the queue is filtered by the
|
|
474
|
+
* caller's own `keep` so the bound is spent on pages that could actually become
|
|
475
|
+
* pages.
|
|
476
|
+
*/
|
|
477
|
+
export async function crawlLinks(entry, opts) {
|
|
478
|
+
const depth = Math.max(0, opts.depth ?? 1);
|
|
479
|
+
const maxVisits = opts.maxVisits ?? 24;
|
|
480
|
+
const found = new Set([entry]);
|
|
481
|
+
const titles = new Map();
|
|
482
|
+
let visited = 0;
|
|
483
|
+
await withBrowser(async (browser) => {
|
|
484
|
+
let frontier = [entry];
|
|
485
|
+
// NAVIGATE ONLY WHILE THE LINKS CAN STILL BE USED. `depth` is how far from
|
|
486
|
+
// the entry a discovered page may be, so the pages AT that distance are
|
|
487
|
+
// results and are never opened: opening them would pay a navigation each for
|
|
488
|
+
// links the bound has already ruled out. The import pass opens them anyway.
|
|
489
|
+
for (let level = 0; level < depth && frontier.length > 0; level += 1) {
|
|
490
|
+
const next = [];
|
|
491
|
+
for (const url of frontier) {
|
|
492
|
+
if (visited >= maxVisits)
|
|
493
|
+
break;
|
|
494
|
+
visited += 1;
|
|
495
|
+
try {
|
|
496
|
+
const got = await readPage(browser, url, 1440, (page) => page.evaluate(linksOnPage));
|
|
497
|
+
if (got.title)
|
|
498
|
+
titles.set(url, got.title);
|
|
499
|
+
// A LEVEL BELOW THE LAST IS WALKED FOR ITS LINKS AND NOT QUEUED: at
|
|
500
|
+
// `depth` the crawl still wants what that page points at, it just
|
|
501
|
+
// must not navigate any further.
|
|
502
|
+
for (const raw of got.links) {
|
|
503
|
+
const url2 = opts.canon(raw);
|
|
504
|
+
if (!url2 || found.has(url2))
|
|
505
|
+
continue;
|
|
506
|
+
found.add(url2);
|
|
507
|
+
next.push(url2);
|
|
508
|
+
}
|
|
509
|
+
}
|
|
510
|
+
catch {
|
|
511
|
+
// A page that will not open contributes nothing and ends nothing.
|
|
512
|
+
}
|
|
513
|
+
}
|
|
514
|
+
frontier = next;
|
|
515
|
+
}
|
|
516
|
+
});
|
|
517
|
+
return { urls: [...found], titles, visited };
|
|
518
|
+
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "sbuilder-mcp",
|
|
3
|
-
"version": "0.
|
|
3
|
+
"version": "0.17.1",
|
|
4
4
|
"description": "MCP server that designs and operates a Store Builder site — pages, data, theme and publish — through the platform's own API and live-edit protocol.",
|
|
5
5
|
"mcpName": "io.github.vuluu2k/sbuilder-mcp",
|
|
6
6
|
"type": "module",
|