sbuilder-mcp 0.20.0 → 0.21.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -6,6 +6,14 @@ All notable changes to this project are documented in this file.
6
6
  The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
7
7
  and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
8
8
 
9
+ ## [0.21.0] - 2026-09-10
10
+
11
+ ### Added
12
+ - sb_import and sb_import_site now detect a `<form>` on the source page and report it as `forms_found` (field count and labels) instead of silently dropping it, pointing the caller at `sb_store action:"form"` to rebuild it with a valid field vocabulary.
13
+
14
+ ### Fixed
15
+ - sb_import_site now rewrites links between the pages it imports so the new site's menu points at itself instead of back at the source it was copied from, and reports any same-origin links left pointing off-site under `links.still_off_site` so the caller knows to raise `max_pages`.
16
+
9
17
  ## [0.20.0] - 2026-09-10
10
18
 
11
19
  ### Added
package/CHANGELOG.vi.md CHANGED
@@ -6,6 +6,14 @@ Mọi thay đổi đáng chú ý của dự án được ghi lại trong file n
6
6
  Định dạng dựa trên [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
7
7
  và dự án tuân theo [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
8
8
 
9
+ ## [0.21.0] - 2026-09-10
10
+
11
+ ### Added
12
+ - sb_import và sb_import_site giờ phát hiện `<form>` trên trang nguồn và báo cáo qua `forms_found` (số field và nhãn) thay vì âm thầm bỏ qua, đồng thời chỉ người gọi tới `sb_store action:"form"` để dựng lại form với vocabulary field hợp lệ.
13
+
14
+ ### Fixed
15
+ - sb_import_site giờ viết lại các link giữa các trang mà nó import, để menu của site mới trỏ về chính nó thay vì trỏ ngược về site nguồn đã copy, và báo cáo các link cùng origin còn trỏ ra ngoài qua `links.still_off_site` để người gọi biết cần tăng `max_pages`.
16
+
9
17
  ## [0.20.0] - 2026-09-10
10
18
 
11
19
  ### Added
@@ -1,5 +1,6 @@
1
1
  import { walk } from '../../core/tree.js';
2
2
  import { stickySeeds } from './sticky.js';
3
+ import { normalizeUrl } from './discover.js';
3
4
  import { ICON_NAMES } from '../../catalog/icons.generated.js';
4
5
  /** Style keys read off a node, ignoring anything unset. */
5
6
  function styleOf(n) {
@@ -416,3 +417,39 @@ export function rehostImages(captured, map) {
416
417
  });
417
418
  return captured.map(visit);
418
419
  }
420
+ /**
421
+ * POINT THE IMPORTED LINKS AT THE IMPORTED PAGES.
422
+ *
423
+ * A captured link keeps the SOURCE's absolute URL, so a site brought over with
424
+ * `sb_import_site` had a menu that sent every visitor back to the website it was
425
+ * copied from — twelve pages built here and not one way to reach any of them.
426
+ * The most basic feature a website has, and the import was quietly working
427
+ * against it.
428
+ *
429
+ * Only the links whose target was ACTUALLY IMPORTED are rewritten. A same-origin
430
+ * link to a page the cap left out is counted rather than pointed at a slug that
431
+ * does not exist here: an off-site link that works beats a local one that 404s,
432
+ * and the count is what tells the caller to raise `max_pages`. A genuinely
433
+ * external link is left alone and is not interesting.
434
+ */
435
+ export function relink(captured, local, origin) {
436
+ let rewritten = 0;
437
+ let unimported = 0;
438
+ const one = (c) => {
439
+ let href = c.href;
440
+ if (href) {
441
+ const norm = normalizeUrl(href);
442
+ const to = norm ? local.get(norm) : undefined;
443
+ if (to) {
444
+ href = to;
445
+ rewritten += 1;
446
+ }
447
+ else if (norm && norm.indexOf(origin) === 0) {
448
+ unimported += 1;
449
+ }
450
+ }
451
+ const kids = c.children ? c.children.map(one) : undefined;
452
+ return { ...c, ...(href ? { href } : {}), ...(kids ? { children: kids } : {}) };
453
+ };
454
+ return { sections: captured.map(one), rewritten, unimported };
455
+ }
@@ -4,7 +4,7 @@ import { capture, captureMany, crawlLinks } from '../vision/capture.js';
4
4
  import { uploadMedia } from '../transport/media.js';
5
5
  import { addSubtree } from '../domains/site/builder.js';
6
6
  import { middleEnd } from '../domains/site/traps.js';
7
- import { toSpecs, tokensFromPage, imageSources, rehostImages, } from '../domains/site/importmap.js';
7
+ import { toSpecs, tokensFromPage, imageSources, rehostImages, relink, } from '../domains/site/importmap.js';
8
8
  import { canonFor, choosePages, normalizeUrl, robotsRules, robotsSitemaps, sitemapUrls, } from '../domains/site/discover.js';
9
9
  import { loadSource } from '../transport/pages.js';
10
10
  import { PageDoc } from '../domains/site/document.js';
@@ -207,6 +207,7 @@ export function registerImportTools(server, ctx, session) {
207
207
  title: shot.title,
208
208
  sections: specs.length,
209
209
  images: images.length,
210
+ ...(shot.forms?.length ? { forms_found: shot.forms } : {}),
210
211
  tokens,
211
212
  skipped: shot.skipped,
212
213
  note: 'Structure and content only — the source\'s CSS and layout are NOT copied, and the ' +
@@ -278,6 +279,7 @@ export function registerImportTools(server, ctx, session) {
278
279
  return text({
279
280
  read: shot.url,
280
281
  added_sections: added,
282
+ ...(shot.forms?.length ? { forms_found: shot.forms } : {}),
281
283
  // WHAT WAS LEFT BEHIND, on the real run too. The dry run said it and the
282
284
  // real one did not, which is the wrong way round: a caller who skipped
283
285
  // the preview is exactly the caller who needs to be told that 21 nodes
@@ -477,6 +479,16 @@ export function registerImportTools(server, ctx, session) {
477
479
  // cannot be — each page is its own create and its own save — so the honest
478
480
  // shape is per-page outcomes. Aborting on the fourth of twelve would leave
479
481
  // three pages built, nine not, and no report saying which.
482
+ // WHERE EACH IMPORTED PAGE WILL LIVE HERE, decided before the first one is
483
+ // built — page two's link to page seven has to work, and page seven does
484
+ // not exist yet. The entry resolves to "/" when it merges into the site's
485
+ // own home page, because that is the address it will answer on.
486
+ const origin = new URL(entry).origin;
487
+ const localPath = new Map();
488
+ for (const p of plan.pages) {
489
+ const isEntry = p.url === entry;
490
+ localPath.set(p.url, isEntry && homepage !== false && home ? '/' : `/${p.slug}`);
491
+ }
480
492
  const built = [];
481
493
  const failed = [];
482
494
  // WHAT EACH PAGE SAYS ITS OWN ADDRESS IS. A sitemap cannot tell you that
@@ -486,6 +498,9 @@ export function registerImportTools(server, ctx, session) {
486
498
  // under two slugs, and nothing in the plan looks wrong.
487
499
  const identities = new Set();
488
500
  const aliased = [];
501
+ let relinked = 0;
502
+ let unimported = 0;
503
+ const formsSeen = [];
489
504
  let lastOpened = '';
490
505
  for (const p of plan.pages) {
491
506
  const shot = byUrl.get(p.url);
@@ -500,8 +515,13 @@ export function registerImportTools(server, ctx, session) {
500
515
  }
501
516
  identities.add(identity);
502
517
  try {
503
- const sections = rehosted.size > 0 ? rehostImages(shot.result.sections, rehosted) : shot.result.sections;
504
- const specs = toSpecs(sections, tokens);
518
+ const hosted = rehosted.size > 0 ? rehostImages(shot.result.sections, rehosted) : shot.result.sections;
519
+ for (const f of shot.result.forms ?? [])
520
+ formsSeen.push({ page: p.slug, ...f });
521
+ const linked = relink(hosted, localPath, origin);
522
+ relinked += linked.rewritten;
523
+ unimported += linked.unimported;
524
+ const specs = toSpecs(linked.sections, tokens);
505
525
  if (specs.length === 0) {
506
526
  failed.push({
507
527
  url: p.url,
@@ -583,6 +603,19 @@ export function registerImportTools(server, ctx, session) {
583
603
  built,
584
604
  ...(failed.length ? { failed } : {}),
585
605
  ...(aliased.length ? { same_page: aliased } : {}),
606
+ links: {
607
+ rewritten: relinked,
608
+ ...(unimported ? { still_off_site: unimported } : {}),
609
+ },
610
+ ...(formsSeen.length
611
+ ? {
612
+ forms_found: formsSeen,
613
+ forms_note: 'A form is NOT a page node here — it lives in its own document, and its fields ' +
614
+ 'are a vocabulary the server validates. sb_store action:"form" seeds any of the ' +
615
+ "platform's 17 templates (contact, subscribe, booking, login, register …) with " +
616
+ 'that document already correct; place the form on the page after.',
617
+ }
618
+ : {}),
586
619
  ...(Object.keys(plan.skipped).length || found.aliases
587
620
  ? { skipped: { ...plan.skipped, ...(found.aliases ? { 'canonical-alias': found.aliases } : {}) } }
588
621
  : {}),
@@ -70,8 +70,13 @@ function capturePage(limits) {
70
70
  // one kind of content that could not survive the trip at all.
71
71
  const IGNORE = new Set([
72
72
  'SCRIPT', 'STYLE', 'NOSCRIPT', 'TEMPLATE', 'SVG', 'PATH', 'CANVAS',
73
- 'NAV', 'FORM', 'INPUT', 'SELECT', 'TEXTAREA', 'BUTTON',
73
+ // FORM is NOT here: its own branch below records it before returning
74
+ // nothing, so a contact page says it had one. The CONTROLS stay ignored —
75
+ // a stray input outside a form is chrome, and the fields of a form that IS
76
+ // reported are counted there rather than walked into.
77
+ 'NAV', 'INPUT', 'SELECT', 'TEXTAREA', 'BUTTON',
74
78
  ]);
79
+ const forms = [];
75
80
  /** The provider and id behind an embed URL, or null if this platform has no element for it. */
76
81
  const embedOf = (raw) => {
77
82
  const u = raw.split('?')[0];
@@ -322,6 +327,23 @@ function capturePage(limits) {
322
327
  return [{ kind: 'accordion', children: [{ kind: 'accordion-item', text: label, children: body }] }];
323
328
  }
324
329
  // A RULE BETWEEN SECTIONS IS A DESIGN DECISION, and it is one node.
330
+ // A FORM IS SEEN AND NOT REBUILT. Its fields are a vocabulary the server
331
+ // validates (`mapTo`), and `sb_store action:"form"` owns that; what this
332
+ // walk can honestly do is say the page had one, so a contact page does not
333
+ // arrive with no way to contact anybody and nothing saying why.
334
+ if (tag === 'FORM') {
335
+ const controls = Array.from(el.querySelectorAll('input, textarea, select'));
336
+ const labels = [];
337
+ for (const l of Array.from(el.querySelectorAll('label'))) {
338
+ const t = clean(l.textContent);
339
+ if (t && labels.length < 10)
340
+ labels.push(t);
341
+ }
342
+ if (controls.length > 0)
343
+ forms.push({ fields: controls.length, labels });
344
+ skip('form');
345
+ return [];
346
+ }
325
347
  if (tag === 'HR') {
326
348
  taken.nodes++;
327
349
  return [{ kind: 'divider' }];
@@ -608,6 +630,7 @@ function capturePage(limits) {
608
630
  url: here,
609
631
  title: clean(document.title),
610
632
  ...(canonical ? { canonical: abs(canonical) } : {}),
633
+ ...(forms.length ? { forms } : {}),
611
634
  sections,
612
635
  skipped,
613
636
  };
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "sbuilder-mcp",
3
- "version": "0.20.0",
3
+ "version": "0.21.0",
4
4
  "description": "MCP server that designs and operates a Store Builder site — pages, data, theme and publish — through the platform's own API and live-edit protocol.",
5
5
  "mcpName": "io.github.vuluu2k/sbuilder-mcp",
6
6
  "type": "module",