@topy-ai/maggie 0.7.35 → 0.7.37

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (30) hide show
  1. package/README-zh-TW.md +45 -2
  2. package/README.md +108 -4
  3. package/bin/maggie.js +76 -19
  4. package/bundled-contracts/maggie-seo/model-policy-v2.schema.json +29 -0
  5. package/bundled-contracts/maggie-service-booking/notification-lifecycle-v1.schema.json +29 -0
  6. package/bundled-contracts/maggiedash/README.md +3 -0
  7. package/bundled-contracts/maggiedash/booking-access-v1.json +29 -0
  8. package/bundled-contracts/maggiedash/booking-customer-surface-v1.json +46 -0
  9. package/bundled-contracts/maggiedash/booking-email-templates-v1.json +38 -0
  10. package/bundled-contracts/maggiedash/booking-host-adapter-v1.json +76 -0
  11. package/bundled-contracts/maggiedash/booking-ops-evidence-v1.json +19 -0
  12. package/bundled-contracts/maggiedash/execution-board.json +1886 -0
  13. package/bundled-contracts/maggiedash/stripe-booking-capabilities-v1.json +68 -0
  14. package/bundled-skills/README.md +1 -0
  15. package/bundled-skills/catalog.json +4 -0
  16. package/bundled-skills/maggie-blog-bootstrap/SKILL.md +6 -0
  17. package/bundled-skills/maggie-booking/SKILL.md +462 -0
  18. package/bundled-skills/maggie-design/SKILL.md +15 -0
  19. package/bundled-skills/maggie-seo-geo/SKILL.md +24 -2
  20. package/bundled-skills/maggie-service-booking/SKILL.md +19 -0
  21. package/bundled-templates/astro-blog/starter/README.md +4 -0
  22. package/bundled-templates/astro-blog/starter/src/styles/global.css +6 -0
  23. package/bundled-tools/clis/maggie.py +36 -2
  24. package/bundled-tools/clis/maggie_booking.py +1080 -0
  25. package/bundled-tools/clis/maggie_dash.py +181 -5
  26. package/bundled-tools/clis/maggie_release.py +24 -0
  27. package/bundled-tools/clis/maggie_service_booking.py +38 -2
  28. package/bundled-tools/clis/site_audit.py +114 -15
  29. package/bundled-tools/runtime/site_baseline.py +11 -1
  30. package/package.json +1 -1
@@ -12,6 +12,7 @@ from __future__ import annotations
12
12
  import argparse
13
13
  import json
14
14
  import os
15
+ import re
15
16
  import shutil
16
17
  import subprocess
17
18
  import sys
@@ -145,9 +146,131 @@ def manifest_root_file_pairs(source: Path, manifest: dict[str, object], root: Pa
145
146
  return pairs
146
147
 
147
148
 
148
- def command_install_diff(source: Path, manifest: dict[str, object], target: Path, compare_dir: Path | None, force: bool) -> dict[str, object]:
149
+ def manifest_workspace_file_pairs(source: Path, manifest: dict[str, object], root: Path) -> list[tuple[Path, Path, Path]]:
150
+ """Return additional installable workspaces such as ./_maggie/booking."""
151
+ pairs: list[tuple[Path, Path, Path]] = []
152
+ for item in manifest.get("workspaces", []):
153
+ if not isinstance(item, dict):
154
+ raise RuntimeError("MaggieDash workspaces entries must be objects")
155
+ source_relative = Path(str(item.get("source", "")))
156
+ target_relative = Path(str(item.get("target", "")))
157
+ if not source_relative.parts or not target_relative.parts:
158
+ raise RuntimeError("MaggieDash workspaces entries require source and target")
159
+ if source_relative.is_absolute() or target_relative.is_absolute() or ".." in source_relative.parts or ".." in target_relative.parts:
160
+ raise RuntimeError("unsafe MaggieDash workspace path")
161
+ source_item = source / source_relative
162
+ if not source_item.is_dir():
163
+ raise RuntimeError(f"MaggieDash workspace source is missing: {source_relative}")
164
+ for path in sorted(source_item.rglob("*")):
165
+ if path.is_file():
166
+ child = path.relative_to(source_item)
167
+ pairs.append((path, root / target_relative / child, source_relative / child))
168
+ return pairs
169
+
170
+
171
+ def detect_host_framework(root: Path, requested: str) -> str | None:
172
+ """Resolve the optional host scaffold without guessing unsupported apps."""
173
+ if requested == "none":
174
+ return None
175
+ if requested != "auto":
176
+ return requested
177
+ package = root / "package.json"
178
+ try:
179
+ value = json.loads(package.read_text(encoding="utf-8"))
180
+ except (OSError, json.JSONDecodeError):
181
+ value = {}
182
+ dependencies = set((value.get("dependencies") or {})) | set((value.get("devDependencies") or {}))
183
+ if "astro" in dependencies or any(root.glob("src/**/*.astro")):
184
+ return "astro"
185
+ return None
186
+
187
+
188
+ ASTRO_MIDDLEWARE_MARKER = "maggie-auto-rewrite-v1"
189
+
190
+
191
+ def merge_astro_middleware(path: Path) -> dict[str, object]:
192
+ """Add the stable Maggie paths to an existing conventional Astro middleware.
193
+
194
+ The installer must preserve host-owned auth/redirect logic. This narrow
195
+ merge only handles the common `onRequest = defineMiddleware((context,
196
+ next) => { ... })` shape and is idempotent. Unknown middleware shapes are
197
+ reported to the caller instead of being rewritten heuristically.
198
+ """
199
+ try:
200
+ content = path.read_text(encoding="utf-8")
201
+ except OSError as error:
202
+ return {"status": "manual", "reason": f"could not read {path}: {error}"}
203
+ if ASTRO_MIDDLEWARE_MARKER in content:
204
+ return {"status": "unchanged", "reason": "Maggie rewrite block already present"}
205
+ required_patterns = ("/_maggie/booking", "/_maggie/login", "/_maggie/register", "/_maggie/reset-password")
206
+ if all(pattern in content for pattern in required_patterns) and "context.rewrite" in content:
207
+ return {"status": "unchanged", "reason": "existing middleware already exposes Maggie paths"}
208
+ match = re.search(
209
+ r"export\s+const\s+onRequest\s*=\s*defineMiddleware\(\s*\(\s*context\s*,\s*next\s*\)\s*=>\s*\{",
210
+ content,
211
+ )
212
+ if not match:
213
+ return {"status": "manual", "reason": "middleware is not the supported defineMiddleware((context, next) => {}) shape"}
214
+ block = r'''
215
+ // maggie-auto-rewrite-v1: generated by `maggie booking install`; keep host auth below.
216
+ const maggieRewrites: Array<[RegExp, (path: string) => string]> = [
217
+ [/^\/_maggie\/login\/?$/, () => "/maggie/login"],
218
+ [/^\/_maggie\/register\/?$/, () => "/maggie/register"],
219
+ [/^\/_maggie\/reset-password\/?$/, () => "/maggie/reset-password"],
220
+ [/^\/_maggie\/booking\/book\/?$/, () => "/maggie/booking/book"],
221
+ [/^\/_maggie\/booking(\/.*)?$/, (path) => "/maggie/booking" + path.replace(/^\/_maggie\/booking/, "")],
222
+ ];
223
+ for (const [pattern, target] of maggieRewrites) {
224
+ if (pattern.test(context.url.pathname)) {
225
+ const destination = new URL(target(context.url.pathname), context.url);
226
+ destination.search = context.url.search;
227
+ return context.rewrite(destination.pathname + destination.search);
228
+ }
229
+ }
230
+ '''
231
+ merged = content[:match.end()] + block + content[match.end():]
232
+ try:
233
+ path.write_text(merged, encoding="utf-8")
234
+ except OSError as error:
235
+ return {"status": "manual", "reason": f"could not update {path}: {error}"}
236
+ return {"status": "updated", "reason": "inserted idempotent Maggie rewrite block"}
237
+
238
+
239
+ def manifest_host_file_pairs(source: Path, manifest: dict[str, object], root: Path, framework: str | None) -> list[tuple[Path, Path, Path]]:
240
+ """Return non-destructive host files for a declared framework adapter."""
241
+ if not framework:
242
+ return []
243
+ bootstrap = manifest.get("hostBootstrap", {})
244
+ if not isinstance(bootstrap, dict):
245
+ return []
246
+ spec = bootstrap.get(framework)
247
+ if not isinstance(spec, dict):
248
+ raise RuntimeError(
249
+ f"MaggieDash distribution at {source} does not provide a {framework} host bootstrap; "
250
+ f"release a source with manifest.hostBootstrap.{framework} or pass --source to a released checkout"
251
+ )
252
+ source_relative = Path(str(spec.get("source", "")))
253
+ target_relative = Path(str(spec.get("target", ".")))
254
+ if not source_relative.parts or source_relative.is_absolute() or ".." in source_relative.parts:
255
+ raise RuntimeError("unsafe MaggieDash host bootstrap source")
256
+ if target_relative.as_posix() not in {".", ""} and (target_relative.is_absolute() or ".." in target_relative.parts):
257
+ raise RuntimeError("unsafe MaggieDash host bootstrap target")
258
+ source_item = source / source_relative
259
+ if not source_item.is_dir():
260
+ raise RuntimeError(f"MaggieDash host bootstrap source is missing: {source_relative}")
261
+ pairs: list[tuple[Path, Path, Path]] = []
262
+ for path in sorted(source_item.rglob("*")):
263
+ if path.is_file():
264
+ child = path.relative_to(source_item)
265
+ pairs.append((path, root / target_relative / child, source_relative / child))
266
+ return pairs
267
+
268
+
269
+ def command_install_diff(source: Path, manifest: dict[str, object], target: Path, compare_dir: Path | None, force: bool, host_pairs: list[tuple[Path, Path, Path]] | None = None, host_framework: str | None = None) -> dict[str, object]:
149
270
  pairs = manifest_file_pairs(source, manifest, target)
150
271
  root_pairs = manifest_root_file_pairs(source, manifest, target.parents[1])
272
+ workspace_pairs = manifest_workspace_file_pairs(source, manifest, target.parents[1])
273
+ host_pairs = host_pairs or []
151
274
  files: list[dict[str, str]] = []
152
275
  source_relative = {relative for _, _, relative in pairs}
153
276
  for source_file, destination, relative in pairs:
@@ -172,6 +295,24 @@ def command_install_diff(source: Path, manifest: dict[str, object], target: Path
172
295
  else:
173
296
  action = "update" if force else "preserve"
174
297
  files.append({"path": str(relative), "action": action, "comparePath": str(compare_file)})
298
+ for source_file, destination, relative in workspace_pairs:
299
+ compare_file = destination
300
+ if not compare_file.exists():
301
+ action = "add"
302
+ elif compare_file.read_bytes() == source_file.read_bytes():
303
+ action = "unchanged"
304
+ else:
305
+ action = "update" if force else "preserve"
306
+ files.append({"path": str(relative), "action": action, "comparePath": str(compare_file)})
307
+ for source_file, destination, relative in host_pairs:
308
+ compare_file = destination
309
+ if not compare_file.exists():
310
+ action = "add"
311
+ elif compare_file.read_bytes() == source_file.read_bytes():
312
+ action = "unchanged"
313
+ else:
314
+ action = "update" if force else "preserve"
315
+ files.append({"path": f"host/{host_framework}/{relative}", "action": action, "comparePath": str(compare_file)})
175
316
  if compare_dir and compare_dir.exists():
176
317
  for existing in sorted(compare_dir.rglob("*")):
177
318
  if not existing.is_file():
@@ -183,13 +324,14 @@ def command_install_diff(source: Path, manifest: dict[str, object], target: Path
183
324
  summary: dict[str, int] = {}
184
325
  for item in files:
185
326
  summary[item["action"]] = summary.get(item["action"], 0) + 1
186
- return {"status": "dry-run", "version": manifest.get("version"), "target": str(target), "compareDir": str(compare_dir) if compare_dir else None, "summary": summary, "files": files}
327
+ return {"status": "dry-run", "version": manifest.get("version"), "target": str(target), "hostFramework": host_framework, "compareDir": str(compare_dir) if compare_dir else None, "summary": summary, "files": files}
187
328
 
188
329
 
189
330
  def command_install(args: argparse.Namespace) -> int:
190
331
  if not args.dry_run and not args.diff:
191
332
  require_confirm(args)
192
333
  root = project_root(args)
334
+ host_framework = detect_host_framework(root, args.host)
193
335
  source_value = args.source or os.environ.get("MAGGIE_DASH_SOURCE") or DEFAULT_DASH_SOURCE
194
336
  ref = args.ref
195
337
  target_relative = Path(args.target)
@@ -222,6 +364,7 @@ def command_install(args: argparse.Namespace) -> int:
222
364
  raise RuntimeError("unsupported MaggieDash distribution schema")
223
365
  if manifest.get("target") != "_maggie/admin" and args.target == "_maggie/admin":
224
366
  raise RuntimeError("MaggieDash manifest target is not ./_maggie/admin")
367
+ host_pairs = manifest_host_file_pairs(source, manifest, root, host_framework)
225
368
  compare_dir = None
226
369
  if args.existing_dir:
227
370
  compare_dir = Path(args.existing_dir).expanduser()
@@ -229,7 +372,7 @@ def command_install(args: argparse.Namespace) -> int:
229
372
  compare_dir = root / compare_dir
230
373
  compare_dir = compare_dir.resolve()
231
374
  if args.dry_run or args.diff:
232
- emit(command_install_diff(source, manifest, target, compare_dir, args.force))
375
+ emit(command_install_diff(source, manifest, target, compare_dir, args.force, host_pairs, host_framework))
233
376
  return 0
234
377
  installed: list[str] = []
235
378
  preserved: list[str] = []
@@ -264,6 +407,32 @@ def command_install(args: argparse.Namespace) -> int:
264
407
  destination.parent.mkdir(parents=True, exist_ok=True)
265
408
  shutil.copy2(source_file, destination)
266
409
  installed.append(str(destination))
410
+ for source_file, destination, _ in manifest_workspace_file_pairs(source, manifest, root):
411
+ if destination.exists() and not args.force:
412
+ preserved.append(str(destination))
413
+ else:
414
+ destination.parent.mkdir(parents=True, exist_ok=True)
415
+ shutil.copy2(source_file, destination)
416
+ installed.append(str(destination))
417
+ host_installed: list[str] = []
418
+ host_preserved: list[str] = []
419
+ host_updated: list[str] = []
420
+ host_warnings: list[str] = []
421
+ for source_file, destination, _ in host_pairs:
422
+ if destination.exists() and not args.force:
423
+ host_preserved.append(str(destination))
424
+ else:
425
+ destination.parent.mkdir(parents=True, exist_ok=True)
426
+ shutil.copy2(source_file, destination)
427
+ host_installed.append(str(destination))
428
+ if host_framework == "astro":
429
+ middleware = root / "src" / "middleware.ts"
430
+ if middleware.exists():
431
+ merge_result = merge_astro_middleware(middleware)
432
+ if merge_result.get("status") == "updated":
433
+ host_updated.append(str(middleware))
434
+ elif merge_result.get("status") == "manual":
435
+ host_warnings.append(str(merge_result.get("reason")))
267
436
  state = {
268
437
  "schemaVersion": "maggiedash-install.v1",
269
438
  "status": "installed",
@@ -274,13 +443,19 @@ def command_install(args: argparse.Namespace) -> int:
274
443
  "target": str(target),
275
444
  "installed": sorted(set(installed)),
276
445
  "preserved": sorted(set(preserved)),
446
+ "hostFramework": host_framework,
447
+ "hostBootstrap": "installed" if host_framework else "skipped",
448
+ "hostInstalled": sorted(set(host_installed)),
449
+ "hostPreserved": sorted(set(host_preserved)),
450
+ "hostUpdated": sorted(set(host_updated)),
451
+ "hostWarnings": sorted(set(host_warnings)),
277
452
  "updatedAt": utc_now(),
278
453
  }
279
454
  state_path = root / ".maggie" / "dash-install.json"
280
455
  state_path.parent.mkdir(parents=True, exist_ok=True)
281
456
  state_path.write_text(json.dumps(state, indent=2, ensure_ascii=False) + "\n", encoding="utf-8")
282
- emit({"status": "installed", "version": manifest.get("version"), "target": str(target), "installed": len(set(installed)), "preserved": len(set(preserved)), "state": str(state_path)})
283
- return 0
457
+ emit({"status": "needs-attention" if host_warnings else "installed", "version": manifest.get("version"), "target": str(target), "hostFramework": host_framework, "hostInstalled": len(set(host_installed)), "hostPreserved": len(set(host_preserved)), "hostUpdated": len(set(host_updated)), "hostWarnings": host_warnings, "installed": len(set(installed)), "preserved": len(set(preserved)), "state": str(state_path)})
458
+ return 1 if host_warnings else 0
284
459
  finally:
285
460
  if temporary is not None:
286
461
  temporary.cleanup()
@@ -630,6 +805,7 @@ def parser() -> argparse.ArgumentParser:
630
805
  install.add_argument("--source", help="local checkout or Git URL; defaults to MaggieDash repository")
631
806
  install.add_argument("--ref", default="main", help="Git ref when installing from a remote source")
632
807
  install.add_argument("--target", default="_maggie/admin", help="installation target relative to the project")
808
+ install.add_argument("--host", choices=["auto", "astro", "none"], default="auto", help="install the matching non-destructive host scaffold (default: auto-detect Astro)")
633
809
  install.add_argument("--force", action="store_true", help="replace existing dashboard files")
634
810
  install.add_argument("--dry-run", action="store_true", help="show the install plan without writing files")
635
811
  install.add_argument("--diff", action="store_true", help="show the install plan and compare with an existing workspace")
@@ -320,6 +320,29 @@ def editorial_gate(project: Path) -> dict:
320
320
  }
321
321
 
322
322
 
323
+ def social_card_gate(project: Path) -> dict:
324
+ """Include the per-page social-card audit whenever a manifest is present."""
325
+ candidates = (
326
+ (project / "docs" / "social-cards-pages.json", "--pages-file"),
327
+ (project / ".maggie" / "social-cards-pages.json", "--pages-file"),
328
+ (project / "docs" / "public-urls.json", "--urls-file"),
329
+ (project / ".maggie" / "public-urls.json", "--urls-file"),
330
+ )
331
+ candidate = next(((path, flag) for path, flag in candidates if path.is_file()), None)
332
+ if candidate is None:
333
+ return {
334
+ "name": "social-card-audit",
335
+ "passed": True,
336
+ "exitCode": 0,
337
+ "result": {"passed": True, "state": "not-configured", "reason": "add docs/social-cards-pages.json or docs/public-urls.json to enable the release-path audit"},
338
+ "stderr": "",
339
+ }
340
+ path, flag = candidate
341
+ result = run_gate("social-card-audit", [sys.executable, str(ROOT / "maggie_social_cards.py"), "audit", flag, str(path)], project)
342
+ result["result"]["manifest"] = str(path.relative_to(project))
343
+ return result
344
+
345
+
323
346
  def runtime_smoke(base_url: str) -> dict:
324
347
  """Check basic public, admin redirect, and unknown-route behaviour."""
325
348
  base = base_url.rstrip("/") + "/"
@@ -366,6 +389,7 @@ def main() -> int:
366
389
  gates.append(changed_surface_gate(project))
367
390
  gates.append(qa_gate(project, args.environment, args.base_url))
368
391
  gates.append(build_gate(project))
392
+ gates.append(social_card_gate(project))
369
393
  for name, command in build_gates(project, args.environment, args.target, not args.skip_compatibility):
370
394
  gates.append(run_gate(name, command, project))
371
395
  if args.analytics_release_gate:
@@ -7,6 +7,7 @@ import xml.etree.ElementTree as ET
7
7
  from datetime import datetime, timezone
8
8
  from html.parser import HTMLParser
9
9
  from pathlib import Path
10
+ from zoneinfo import ZoneInfo, ZoneInfoNotFoundError
10
11
  from urllib.parse import quote, urljoin, urlsplit, urlparse
11
12
  from urllib.request import Request, urlopen
12
13
 
@@ -20,6 +21,39 @@ DURATION = re.compile(r"(\d{2,3})\s*(?:min|mins|minutes?)", re.I)
20
21
  SUPPLY_STATES = {"live", "withdrawn"}
21
22
  DISPLAY_STATES = {"published", "hidden", "retired"}
22
23
 
24
+
25
+ def parse_booking_time(value, venue_timezone=None, venue_time_fold=None):
26
+ """Parse an appointment time using venue wall-clock rules, not server TZ."""
27
+ raw = str(value or "").strip()
28
+ parsed = datetime.fromisoformat(raw.replace("Z", "+00:00"))
29
+ if parsed.tzinfo is None:
30
+ if not venue_timezone:
31
+ raise ValueError("venueTimezone is required for a timezone-naive booking time")
32
+ try:
33
+ zone = ZoneInfo(str(venue_timezone))
34
+ except ZoneInfoNotFoundError as exc:
35
+ raise ValueError("venueTimezone is not an installed IANA timezone") from exc
36
+ candidates = []
37
+ for fold in (0, 1):
38
+ candidate = parsed.replace(tzinfo=zone, fold=fold)
39
+ round_trip = candidate.astimezone(timezone.utc).astimezone(zone).replace(tzinfo=None)
40
+ if round_trip == parsed:
41
+ candidates.append(candidate)
42
+ if not candidates:
43
+ raise ValueError("venue wall-clock time does not exist because of a DST transition")
44
+ if len({candidate.utcoffset() for candidate in candidates}) > 1:
45
+ if venue_time_fold not in (0, 1):
46
+ raise ValueError("venueTimeFold 0 or 1 is required for an ambiguous DST time")
47
+ parsed = parsed.replace(tzinfo=zone, fold=venue_time_fold)
48
+ else:
49
+ parsed = candidates[0]
50
+ elif venue_timezone:
51
+ try:
52
+ parsed = parsed.astimezone(ZoneInfo(str(venue_timezone)))
53
+ except ZoneInfoNotFoundError as exc:
54
+ raise ValueError("venueTimezone is not an installed IANA timezone") from exc
55
+ return parsed
56
+
23
57
  def slug(value):
24
58
  value = re.sub(r"[^a-z0-9]+", "-", value.lower()).strip("-")
25
59
  return value or "service"
@@ -1081,8 +1115,10 @@ def cmd_variant_audit(args):
1081
1115
  if typ=="event" and not v.get("eventId"): errors.append(f"{prefix}: eventId is required")
1082
1116
  if typ=="promotion" and not v.get("promotionId"): errors.append(f"{prefix}: promotionId is required")
1083
1117
  if typ in {"holiday","seasonal","event","promotion"}:
1084
- try: start=datetime.fromisoformat(str(v.get("startsAt","")).replace("Z","+00:00")); end=datetime.fromisoformat(str(v.get("endsAt","")).replace("Z","+00:00"))
1085
- except ValueError: errors.append(f"{prefix}: startsAt and endsAt are required ISO timestamps"); start=end=None
1118
+ venue_timezone = v.get("venueTimezone") or (payload.get("venueTimezone") if isinstance(payload, dict) else None)
1119
+ fold = v.get("venueTimeFold") if "venueTimeFold" in v else (payload.get("venueTimeFold") if isinstance(payload, dict) else None)
1120
+ try: start=parse_booking_time(v.get("startsAt"), venue_timezone, fold); end=parse_booking_time(v.get("endsAt"), venue_timezone, fold)
1121
+ except ValueError as exc: errors.append(f"{prefix}: startsAt and endsAt require venue wall-clock timezone ({exc})"); start=end=None
1086
1122
  if start and end and end<=start: errors.append(f"{prefix}: endsAt must be after startsAt")
1087
1123
  if end and end<=as_of and v.get("indexable") is True: errors.append(f"{prefix}: expired variant cannot be indexable")
1088
1124
  indexable=v.get("indexable") is True; noindex=v.get("noIndex") is True
@@ -26,7 +26,10 @@ def stable_structure_reference(key: str, value: str) -> str:
26
26
  path = parsed.path
27
27
  if not re.search(r"\.(?:css|js|mjs|png|jpe?g|gif|webp|avif|svg|ico|woff2?|ttf|otf)$", path, re.I):
28
28
  return value
29
- path = re.sub(r"(?:[-_.])[0-9a-f]{8,64}(?=\.[^.]+$)", "-asset", path, flags=re.I)
29
+ # Vite/Astro and similar bundlers also emit base64url-safe digests such as
30
+ # ``SiteLayout.Bdp0itNn.css``. Restrict this to a digest-looking token
31
+ # immediately before a known asset extension.
32
+ path = re.sub(r"(?:[-_.])[A-Za-z0-9_-]{8,64}(?=\.[^.]+$)", "-asset", path)
30
33
  return urlunparse((parsed.scheme, parsed.netloc, path, parsed.params, "", ""))
31
34
 
32
35
 
@@ -65,9 +68,14 @@ class PageParser(HTMLParser):
65
68
  self.headings = []
66
69
  self.paragraphs = []
67
70
  self.links = []
71
+ self.fragment_ids = set()
68
72
 
69
73
  def handle_starttag(self, tag, attrs):
70
74
  data = dict(attrs)
75
+ if data.get("id"):
76
+ self.fragment_ids.add(data["id"])
77
+ if tag == "a" and data.get("name"):
78
+ self.fragment_ids.add(data["name"])
71
79
  if data.get("data-template"):
72
80
  self.templates.add(data["data-template"])
73
81
  if tag in {"script", "style", "noscript"}:
@@ -184,6 +192,7 @@ def audit_page(url: str, html: str, status: int, content_type: str, expected_lan
184
192
  "title": page.title, "meta": page.meta, "canonical": canonical,
185
193
  "lang": page.lang, "hreflang": page.hreflang, "robots": robots,
186
194
  "jsonld": page.jsonld_values, "images": page.images,
195
+ "fragmentIds": sorted(page.fragment_ids),
187
196
  "structureHash": hashlib.sha256(json.dumps(stable_structure(page.structure), sort_keys=True).encode()).hexdigest(),
188
197
  "textHash": hashlib.sha256(" ".join(page.visible_text).encode()).hexdigest(),
189
198
  },
@@ -298,6 +307,99 @@ def access_log_sitemap_check(path: Path, sitemap_path: str = "/sitemap.xml") ->
298
307
  return {"provided": True, "path": str(path), "requestCount": len(requests), "googlebotRequestCount": len(googlebot), "status": "requested" if requests else "not-requested", "diagnosis": "googlebot-request-observed" if googlebot else ("other-client-request-only" if requests else "no-matching-request")}
299
308
 
300
309
 
310
+ def internal_link_check(base: str, pages: list[dict], fetcher=fetch) -> dict[str, object]:
311
+ """Validate same-origin links and fragment targets discovered during a crawl."""
312
+ origin = urlparse(base)
313
+ page_by_url = {page.get("url"): page for page in pages if page.get("url")}
314
+ errors: list[dict[str, str]] = []
315
+ checked = 0
316
+ seen: set[tuple[str, str]] = set()
317
+ for page in pages:
318
+ page_url = str(page.get("url") or "")
319
+ for link in (page.get("contentContract") or {}).get("links", []):
320
+ target = urljoin(page_url, str(link.get("href") or ""))
321
+ parsed = urlparse(target)
322
+ if parsed.scheme not in {"http", "https"} or (parsed.scheme, parsed.netloc.lower()) != (origin.scheme, origin.netloc.lower()):
323
+ continue
324
+ fragment = parsed.fragment
325
+ target_url = urlunparse((parsed.scheme, parsed.netloc, parsed.path or "/", parsed.params, parsed.query, ""))
326
+ key = (target_url, fragment)
327
+ if key in seen:
328
+ continue
329
+ seen.add(key)
330
+ checked += 1
331
+ target_page = page_by_url.get(target_url)
332
+ if target_page and isinstance(target_page.get("contract"), dict):
333
+ status = target_page.get("status")
334
+ content_type = target_page.get("content_type")
335
+ fragment_ids = set(target_page["contract"].get("fragmentIds", []))
336
+ else:
337
+ try:
338
+ status, content_type, body = fetcher(target_url)
339
+ parsed_target = PageParser()
340
+ parsed_target.feed(body)
341
+ fragment_ids = parsed_target.fragment_ids
342
+ except Exception as exc:
343
+ errors.append({"source": page_url, "url": target_url, "reason": type(exc).__name__})
344
+ continue
345
+ if status != 200 or content_type != "text/html":
346
+ errors.append({"source": page_url, "url": target_url, "reason": "target did not return HTTP 200 HTML"})
347
+ elif fragment and fragment not in fragment_ids:
348
+ errors.append({"source": page_url, "url": target, "reason": "fragment target not found"})
349
+ return {"ok": not errors, "linksChecked": checked, "errors": errors}
350
+
351
+
352
+ def mark_unstable_pages(pages: list[dict]) -> None:
353
+ """Refetch baseline candidates once and label per-response markup drift."""
354
+ for page in pages:
355
+ if not isinstance(page.get("contract"), dict):
356
+ continue
357
+ try:
358
+ status, content_type, body = fetch(page["url"])
359
+ second = audit_page(page["url"], body, status, content_type)
360
+ first_contract = page["contract"]
361
+ second_contract = second["contract"]
362
+ first_content = page.get("contentContract", {})
363
+ second_content = second.get("contentContract", {})
364
+ changed = any(first_contract.get(key) != second_contract.get(key) for key in ("structureHash", "textHash")) or first_content.get("contentHash") != second_content.get("contentHash")
365
+ page["stability"] = {
366
+ "status": "unstable" if changed else "stable",
367
+ "changedBetweenFetches": changed,
368
+ "secondStructureHash": second_contract.get("structureHash"),
369
+ "secondTextHash": second_contract.get("textHash"),
370
+ }
371
+ except Exception as exc:
372
+ page["stability"] = {"status": "recapture-failed", "changedBetweenFetches": False, "error": type(exc).__name__}
373
+
374
+
375
+ def parse_baseline_reasons(values: list[str], drift_urls: set[str], reason_all: str | None = None) -> dict[str, str]:
376
+ """Parse per-URL and site-wide reviewer reasons with actionable errors."""
377
+ reasons: dict[str, str] = {}
378
+ default = reason_all.strip() if reason_all and reason_all.strip() else ""
379
+ for value in values:
380
+ url, separator, reason = value.partition("=")
381
+ if not separator or not url.strip() or not reason.strip():
382
+ raise ValueError("each --reason must use URL=REASON")
383
+ url, reason = url.strip(), reason.strip()
384
+ if url == "*":
385
+ if default:
386
+ raise ValueError("duplicate site-wide baseline change reason")
387
+ default = reason
388
+ continue
389
+ if url in reasons:
390
+ raise ValueError("duplicate baseline change reason: " + url)
391
+ if url not in drift_urls:
392
+ raise ValueError("baseline change reason references a URL without a detected drift: " + url)
393
+ reasons[url] = reason
394
+ if default:
395
+ for url in drift_urls:
396
+ reasons.setdefault(url, default)
397
+ missing_reasons = sorted(drift_urls - reasons.keys())
398
+ if missing_reasons:
399
+ raise ValueError("every drifting URL requires a reviewer reason: " + ", ".join(missing_reasons))
400
+ return reasons
401
+
402
+
301
403
  def main() -> int:
302
404
  parser = argparse.ArgumentParser()
303
405
  parser.add_argument("url")
@@ -309,6 +411,7 @@ def main() -> int:
309
411
  parser.add_argument("--languages", help="comma-separated expected languages/locales, e.g. en-GB,es-MX,ja-JP")
310
412
  parser.add_argument("--markets", help="comma-separated markets: global,uk,us")
311
413
  parser.add_argument("--check-hreflang", action="store_true")
414
+ parser.add_argument("--check-internal-links", action="store_true", help="validate same-origin links and fragment targets during a crawl")
312
415
  parser.add_argument("--check-translation-completeness", action="store_true")
313
416
  parser.add_argument("--access-log", type=Path, help="optional local access log for crawler-request diagnosis")
314
417
  parser.add_argument("--require-sitemap-request", action="store_true", help="fail unless the supplied access log contains a sitemap request")
@@ -319,12 +422,15 @@ def main() -> int:
319
422
  parser.add_argument("--recapture-baseline", type=Path, help="write a reviewed versioned baseline after reviewing every drift")
320
423
  parser.add_argument("--baseline-id", help="required stable ID for --recapture-baseline")
321
424
  parser.add_argument("--reason", action="append", default=[], metavar="URL=REASON", help="reviewer-approved reason for one drifting URL; repeat for every drift")
425
+ parser.add_argument("--reason-all", help="reviewer-approved reason applied to every drifting URL during baseline recapture")
322
426
  parser.add_argument("--reviewer", help="required for --save-baseline")
323
427
  args = parser.parse_args()
324
428
  if args.max_pages < 1:
325
429
  parser.error("max-pages must be positive")
326
430
  if (args.save_baseline or args.baseline) and not args.crawl:
327
431
  parser.error("baseline operations require --crawl")
432
+ if args.check_internal_links and not args.crawl:
433
+ parser.error("internal link checks require --crawl")
328
434
  if args.save_baseline and not (args.reviewer or "").strip():
329
435
  parser.error("save-baseline requires --reviewer")
330
436
  if args.recapture_baseline:
@@ -422,6 +528,9 @@ def main() -> int:
422
528
  crawl["sitemapOriginPolicy"] = "declared-origin" if args.preserve_sitemap_origin else "requested-origin"
423
529
  crawl["passed"] = crawl["complete"] and bool(urls) and len(crawl["pages"]) == len(urls) and all(item["passed"] for item in crawl["pages"])
424
530
  crawl["url_count"] = len(urls)
531
+ if args.check_internal_links:
532
+ crawl["internal_links"] = internal_link_check(base, crawl["pages"])
533
+ crawl["passed"] = crawl["passed"] and crawl["internal_links"]["ok"]
425
534
  except Exception as exc:
426
535
  crawl = {"enabled": True, "passed": False, "error": type(exc).__name__, "pages": []}
427
536
 
@@ -431,6 +540,8 @@ def main() -> int:
431
540
  try:
432
541
  current_passed = result["passed"]
433
542
  if args.baseline:
543
+ if args.crawl:
544
+ mark_unstable_pages(crawl["pages"])
434
545
  previous_baseline = json.loads(args.baseline.read_text())
435
546
  result["baseline"] = site_baseline.compare(previous_baseline, result)
436
547
  result["passed"] = result["passed"] and result["baseline"]["passed"]
@@ -439,20 +550,8 @@ def main() -> int:
439
550
  drift_urls = set(comparison.get("added", [])) | set(comparison.get("removed", []))
440
551
  drift_urls |= {item["url"] for item in comparison.get("changed", []) if isinstance(item, dict) and item.get("url")}
441
552
  drift_urls |= {item["url"] for item in comparison.get("contentChanged", []) if isinstance(item, dict) and item.get("url")}
442
- reasons: dict[str, str] = {}
443
- for value in args.reason:
444
- url, separator, reason = value.partition("=")
445
- if not separator or not url.strip() or not reason.strip():
446
- raise ValueError("each --reason must use URL=REASON")
447
- url = url.strip()
448
- if url in reasons:
449
- raise ValueError("duplicate baseline change reason")
450
- if url not in drift_urls:
451
- raise ValueError("baseline change reason references a URL without a detected drift")
452
- reasons[url] = reason.strip()
453
- missing_reasons = sorted(drift_urls - reasons.keys())
454
- if missing_reasons:
455
- raise ValueError("every drifting URL requires a reviewer reason")
553
+ drift_urls |= set(comparison.get("unstable", []))
554
+ reasons = parse_baseline_reasons(args.reason, drift_urls, args.reason_all)
456
555
  if not current_passed:
457
556
  raise ValueError("baseline recapture requires a passing current crawl")
458
557
  recapture_report = dict(result)
@@ -36,6 +36,8 @@ def snapshot(
36
36
  for page in pages:
37
37
  if not page.get("passed") or not isinstance(page.get("contract"), dict):
38
38
  raise ValueError("baseline requires successful page contracts")
39
+ if (page.get("stability") or {}).get("status") == "unstable":
40
+ raise ValueError("baseline requires stable page contracts")
39
41
  if page["url"] in contracts:
40
42
  raise ValueError("duplicate page URL")
41
43
  if urlparse(page["url"]).query:
@@ -75,11 +77,16 @@ def compare(baseline: dict, report: dict) -> dict:
75
77
  expected = baseline["pages"]
76
78
  added, removed = sorted(current.keys() - expected.keys()), sorted(expected.keys() - current.keys())
77
79
  changes = []
80
+ unstable = []
78
81
  for url in sorted(expected.keys() & current.keys()):
79
82
  actual = current[url]
80
83
  if not isinstance(actual, dict):
81
84
  errors.append("missing page contract: " + url)
82
85
  continue
86
+ page = next((item for item in crawl_pages if item.get("url") == url), {})
87
+ if (page.get("stability") or {}).get("status") == "unstable":
88
+ unstable.append(url)
89
+ continue
83
90
  fields = sorted(key for key in expected[url].keys() | actual.keys() if expected[url].get(key) != actual.get(key))
84
91
  if fields:
85
92
  changes.append({"url": url, "fields": fields})
@@ -88,6 +95,8 @@ def compare(baseline: dict, report: dict) -> dict:
88
95
  content_not_compared = []
89
96
  if expected_content:
90
97
  for url in sorted(expected_content.keys() & current_content.keys()):
98
+ if url in unstable:
99
+ continue
91
100
  expected_contract = expected_content[url]
92
101
  current_contract = current_content[url]
93
102
  ignored_legacy_fields = {"images", "imageAlts"} if "images" not in expected_contract else set()
@@ -99,9 +108,10 @@ def compare(baseline: dict, report: dict) -> dict:
99
108
  missing_content = sorted(expected_content.keys() - current_content.keys())
100
109
  if missing_content:
101
110
  errors.append("missing content contract: " + ", ".join(missing_content))
102
- return {"passed": not errors and not added and not removed and not changes and not content_changes,
111
+ return {"passed": not errors and not added and not removed and not changes and not content_changes and not unstable,
103
112
  "errors": errors, "added": added, "removed": removed, "changed": changes,
104
113
  "contentChanged": content_changes,
114
+ "unstable": unstable,
105
115
  "contentNotCompared": content_not_compared,
106
116
  "excludedQueryUrls": query_urls}
107
117
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@topy-ai/maggie",
3
- "version": "0.7.35",
3
+ "version": "0.7.37",
4
4
  "description": "Install and manage Maggie Skills for AI coding agents",
5
5
  "license": "MIT",
6
6
  "type": "module",