@topy-ai/maggie 0.5.0 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -46,7 +46,7 @@ Then invoke the installed skills from your coding agent, for example:
46
46
 
47
47
  Maggie keeps the existing project foundation and asks for decisions before
48
48
  shared routes, analytics, or publishing boundaries change. The current
49
- package ships 16 installable skills and a local-first MaggieDash foundation.
49
+ package ships 18 installable skills and a local-first MaggieDash foundation.
50
50
 
51
51
  ## Command surface
52
52
 
@@ -66,6 +66,9 @@ maggie memory ... # confirmed preferences and lessons
66
66
  maggie feedback ... # redact, preview, submit, list
67
67
  maggie localization ... # plan, validate, review, publish, stale
68
68
  maggie service ... # import, sync, generate, validate
69
+ maggie seo performance ... # sampled PageSpeed/CWV report and baseline
70
+ maggie seo images ... # inventory, variants, confirmation, validate
71
+ maggie seo sitemap ... # typed plan, validate, apply, rollback
69
72
  maggie deployment | migration | release | analytics | schedule
70
73
  maggie api lifecycle | site-audit | ops audit
71
74
  ```
@@ -77,11 +80,38 @@ history only and is not an installable package skill.
77
80
  with `--confirm`, removes only known retired EmDash artifacts. It preserves
78
81
  `docs/de-emdash-*` migration history and `.maggie` marketplace state.
79
82
 
83
+ ### SEO performance, images, and sitemaps
84
+
85
+ The SEO workflow keeps performance evidence separate from deterministic HTML
86
+ correctness checks. Performance reports use multiple samples, exclude duplicate
87
+ PageSpeed `fetchTime` observations, and report median/range before comparing a
88
+ baseline:
89
+
90
+ ```bash
91
+ maggie seo performance --url https://example.com --strategy mobile,desktop \
92
+ --samples 3 --spacing-seconds 30 --output docs/performance-report.json
93
+ ```
94
+
95
+ Responsive image planning emits only confirmed candidates in `srcset`, while
96
+ sitemap planning groups content types, emits valid empty chunk 1 files, and
97
+ rejects relative or off-origin `<loc>` values:
98
+
99
+ ```bash
100
+ maggie seo images inventory --project . --output docs/image-inventory.json
101
+ maggie seo images plan --inventory docs/image-inventory.json --output docs/image-variants.json
102
+ maggie seo sitemap plan --origin https://example.com --routes-file docs/routes.tsv \
103
+ --output docs/sitemap-plan.json
104
+ ```
105
+
106
+ Read the [performance PRD](https://github.com/TOPY-AI-LTD/ai-cmo-skills/blob/main/docs/seo-performance-cwv-prd.md)
107
+ and [image/sitemap PRD](https://github.com/TOPY-AI-LTD/ai-cmo-skills/blob/main/docs/image-sitemap-structure-prd.md)
108
+ for adapter, privacy, apply, rollback, and release gates.
109
+
80
110
  Recommended upgrade sequence for the current release:
81
111
 
82
112
  ```bash
83
- npx @topy-ai/maggie@0.2.9 update --project . --force
84
- npx @topy-ai/maggie@0.2.9 cleanup --project .
113
+ npx @topy-ai/maggie@0.6.0 update --project . --force
114
+ npx @topy-ai/maggie@0.6.0 cleanup --project .
85
115
  ```
86
116
 
87
117
  ## MaggieDash lifecycle
package/bin/maggie.js CHANGED
@@ -109,6 +109,7 @@ Usage:
109
109
  maggie api lifecycle --project PATH [--execute --allow-quota]
110
110
  maggie memory <init|list|search|context|add|record-error|transition|export> --project PATH
111
111
  maggie localization <plan|preview|validate|review|publish|stale|glossary> [options]
112
+ maggie seo performance|images|sitemap [options]
112
113
  maggie feedback <collect|preview|submit|list> [options]
113
114
  maggie site-audit URL [--crawl] [--languages en-GB,es-MX,ja-JP] [--check-hreflang]
114
115
  maggie ops audit --project PATH
@@ -208,7 +209,11 @@ function install(args) {
208
209
  const tools = join(root, "tools");
209
210
  if (existsSync(TOOLS_ROOT)) {
210
211
  for (const group of ["clis", "runtime", "integrations"]) {
211
- if (existsSync(join(TOOLS_ROOT, group))) copyIfMissing(join(TOOLS_ROOT, group), join(tools, group), force);
212
+ if (existsSync(join(TOOLS_ROOT, group))) {
213
+ const target = join(tools, group);
214
+ if (existsSync(target)) syncTree(join(TOOLS_ROOT, group), target, force);
215
+ else copyIfMissing(join(TOOLS_ROOT, group), target, force);
216
+ }
212
217
  }
213
218
  }
214
219
  if (existsSync(join(DESIGN_ROOT, "SPA-DESIGN.md"))) {
@@ -244,7 +249,10 @@ function update(args) {
244
249
  if (existsSync(REFERENCES_ROOT) && existsSync(join(agentRoot, "references"))) updated += syncTree(REFERENCES_ROOT, join(agentRoot, "references"), force);
245
250
  }
246
251
  const tools = join(root, "tools");
247
- if (existsSync(TOOLS_ROOT)) for (const group of ["clis", "integrations"]) if (existsSync(join(tools, group))) updated += syncTree(join(TOOLS_ROOT, group), join(tools, group), force);
252
+ if (existsSync(TOOLS_ROOT)) for (const group of ["clis", "runtime", "integrations"]) if (existsSync(join(TOOLS_ROOT, group))) {
253
+ const target = join(tools, group);
254
+ if (existsSync(target)) updated += syncTree(join(TOOLS_ROOT, group), target, force);
255
+ }
248
256
  if (existsSync(join(DESIGN_ROOT, "SPA-DESIGN.md")) && existsSync(join(root, ".maggie", "design-reference"))) {
249
257
  updated += syncTree(DESIGN_ROOT, join(root, ".maggie", "design-reference"), force);
250
258
  }
@@ -282,6 +290,13 @@ function service(args) {
282
290
  process.exitCode = result.status ?? 1;
283
291
  }
284
292
 
293
+ function seo(args) {
294
+ const command = args[0];
295
+ const scripts = { performance: "maggie_performance.py", images: "maggie_images.py", sitemap: "maggie_sitemap.py" };
296
+ if (!scripts[command]) throw new Error("seo command must be performance, images, or sitemap");
297
+ workflowCli(scripts[command], args.slice(1));
298
+ }
299
+
285
300
  function workflowCli(name, args) {
286
301
  const root = projectRoot(args);
287
302
  const script = join(root, "tools", "clis", name);
@@ -358,6 +373,7 @@ try {
358
373
  else if (command === "auth") workflowCli("maggie_auth.py", args);
359
374
  else if (command === "blog") workflowCli("maggie_blog.py", args);
360
375
  else if (command === "service") service(args);
376
+ else if (command === "seo") seo(args);
361
377
  else if (command === "ops") workflowCli("maggie_ops.py", args);
362
378
  else if (command === "deployment") workflowCli("maggie_deployment.py", args);
363
379
  else if (command === "migration") workflowCli("maggie_migration.py", args);
@@ -0,0 +1,14 @@
1
+ {
2
+ "$schema": "https://json-schema.org/draft/2020-12/schema",
3
+ "$id": "https://schemas.noblox.app/maggie-seo/image-variants-v1.schema.json",
4
+ "title": "Maggie image variants manifest",
5
+ "type": "object",
6
+ "required": ["schemaVersion", "manifestId", "createdAt", "assets", "provenance"],
7
+ "properties": {
8
+ "schemaVersion": {"const": "maggie-seo-image-variants.v1"},
9
+ "manifestId": {"type": "string", "pattern": "^images-[A-Za-z0-9_-]+$"},
10
+ "createdAt": {"type": "string", "format": "date-time"},
11
+ "assets": {"type": "array", "items": {"type": "object", "required": ["assetId", "source", "variants", "markupPolicy"]}},
12
+ "provenance": {"type": "object", "required": ["redactions"]}
13
+ }
14
+ }
@@ -0,0 +1,46 @@
1
+ {
2
+ "$schema": "https://json-schema.org/draft/2020-12/schema",
3
+ "$id": "https://schemas.noblox.app/maggie-seo/performance-report-v1.schema.json",
4
+ "title": "Maggie SEO performance report",
5
+ "type": "object",
6
+ "required": ["schemaVersion", "reportId", "createdAt", "provider", "policy", "pages", "summary", "provenance"],
7
+ "properties": {
8
+ "schemaVersion": {"const": "maggie-seo-performance.v1"},
9
+ "reportId": {"type": "string", "pattern": "^perf-[A-Za-z0-9_-]+$"},
10
+ "createdAt": {"type": "string", "format": "date-time"},
11
+ "provider": {"const": "pagespeed-insights"},
12
+ "providerApiVersion": {"type": "string"},
13
+ "project": {"type": "object", "additionalProperties": true},
14
+ "policy": {"$ref": "#/$defs/policy"},
15
+ "pages": {"type": "array", "items": {"$ref": "#/$defs/page"}},
16
+ "summary": {"type": "object", "required": ["status"], "properties": {"status": {"enum": ["pass", "fail", "inconclusive"]}}, "additionalProperties": true},
17
+ "provenance": {"type": "object", "required": ["commands", "redactions"], "additionalProperties": true}
18
+ },
19
+ "$defs": {
20
+ "policy": {
21
+ "type": "object",
22
+ "required": ["strategies", "requestedSamples", "minimumIndependentSamples", "spacingSeconds", "duplicateFetchTimePolicy"],
23
+ "properties": {
24
+ "strategies": {"type": "array", "items": {"enum": ["desktop", "mobile"]}},
25
+ "requestedSamples": {"type": "integer", "minimum": 1},
26
+ "minimumIndependentSamples": {"type": "integer", "minimum": 1},
27
+ "spacingSeconds": {"type": "number", "minimum": 0},
28
+ "duplicateFetchTimePolicy": {"const": "exclude"}
29
+ },
30
+ "additionalProperties": true
31
+ },
32
+ "page": {
33
+ "type": "object",
34
+ "required": ["requestedUrl", "strategy", "samples", "aggregate", "fieldData", "comparison"],
35
+ "properties": {
36
+ "requestedUrl": {"type": "string", "format": "uri"},
37
+ "strategy": {"enum": ["desktop", "mobile"]},
38
+ "samples": {"type": "array"},
39
+ "aggregate": {"type": "object"},
40
+ "fieldData": {"type": "object"},
41
+ "comparison": {"type": "object"}
42
+ },
43
+ "additionalProperties": true
44
+ }
45
+ }
46
+ }
@@ -0,0 +1,16 @@
1
+ {
2
+ "$schema": "https://json-schema.org/draft/2020-12/schema",
3
+ "$id": "https://schemas.noblox.app/maggie-seo/sitemap-plan-v1.schema.json",
4
+ "title": "Maggie structured sitemap plan",
5
+ "type": "object",
6
+ "required": ["schemaVersion", "planId", "origin", "groups", "index", "redirects", "validation"],
7
+ "properties": {
8
+ "schemaVersion": {"const": "maggie-seo-sitemap-plan.v1"},
9
+ "planId": {"type": "string", "pattern": "^sitemap-[A-Za-z0-9_-]+$"},
10
+ "origin": {"type": "string", "format": "uri"},
11
+ "groups": {"type": "array"},
12
+ "index": {"type": "object"},
13
+ "redirects": {"type": "array"},
14
+ "validation": {"type": "object", "required": ["status", "errors", "warnings"]}
15
+ }
16
+ }
@@ -19,6 +19,39 @@ voice, CTA alignment, and observed search/AI outcomes.
19
19
 
20
20
  ## Choose the mode
21
21
 
22
+ For sampled PageSpeed performance and Core Web Vitals evidence, run:
23
+
24
+ ```bash
25
+ maggie seo performance --url https://example.com --strategy mobile,desktop \
26
+ --samples 3 --spacing-seconds 30 --output docs/performance-report.json
27
+ ```
28
+
29
+ The performance workflow excludes duplicate provider `fetchTime` observations,
30
+ reports median/range, and marks a baseline comparison `inconclusive` when it
31
+ does not have enough independent samples. `PAGESPEED_API_KEY` is optional for
32
+ the provider but is never written to reports. Read the [performance PRD](https://github.com/TOPY-AI-LTD/ai-cmo-skills/blob/main/docs/seo-performance-cwv-prd.md)
33
+ when selecting thresholds or adding a provider.
34
+
35
+ For responsive image planning and typed sitemap planning, run:
36
+
37
+ ```bash
38
+ maggie seo images inventory --project . --output docs/image-inventory.json
39
+ maggie seo images plan --inventory docs/image-inventory.json --output docs/image-variants.json
40
+ maggie seo images validate --manifest docs/image-variants.json
41
+ maggie seo sitemap plan --origin https://example.com --routes-file docs/routes.tsv \
42
+ --output docs/sitemap-plan.json
43
+ maggie seo sitemap validate --plan docs/sitemap-plan.json
44
+ # After an approved host adapter apply, rollback uses its exact backup manifest.
45
+ maggie seo sitemap rollback --backup-manifest .maggie-sitemap-backups/<plan>/backup-manifest.json \
46
+ --public-dir public --confirm
47
+ ```
48
+
49
+ Only confirmed image variants may enter `srcset`; `apply` requires an explicit
50
+ confirmation and host adapter. Sitemap plans keep content types separate,
51
+ emit chunk 1 for empty enabled types, enforce absolute same-origin URLs, and
52
+ record redirects for removed sitemap files. Read the [image and sitemap PRD](https://github.com/TOPY-AI-LTD/ai-cmo-skills/blob/main/docs/image-sitemap-structure-prd.md)
53
+ for adapter and rollback rules.
54
+
22
55
  For a deterministic technical smoke check, run:
23
56
 
24
57
  ```bash
@@ -30,7 +63,10 @@ interpretation, and content creation remain agent-reviewed decisions.
30
63
  With `--crawl`, it resolves sitemap indexes and audits every listed HTML route
31
64
  for HTTP 200, one H1, title/description, locale, canonical, Open Graph/Twitter
32
65
  metadata, valid entity JSON-LD, and image alt text. Non-HTML endpoints such as
33
- RSS must not be included in the HTML page sitemap.
66
+ RSS must not be included in the HTML page sitemap. Sitemap `<loc>` values must
67
+ be absolute HTTP(S) URLs; the audit reports relative values as a failure even
68
+ when it can resolve them for continued crawling, so one malformed entry cannot
69
+ hide a protocol violation.
34
70
  Use `--output <project>/docs/site-audit.json` to persist the evidence used by
35
71
  the release review.
36
72
 
@@ -0,0 +1,87 @@
1
+ #!/usr/bin/env python3
2
+ """Inventory and validate safe responsive image variant manifests."""
3
+
4
+ from __future__ import annotations
5
+
6
+ import argparse
7
+ import json
8
+ import sys
9
+ from pathlib import Path
10
+ from urllib.request import Request, urlopen
11
+
12
+ sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "runtime"))
13
+ from maggie_media import confirm_variants, inventory_project, plan_variants, validate_manifest # noqa: E402
14
+
15
+
16
+ def write_json(path: str, value: dict) -> None:
17
+ target = Path(path)
18
+ target.parent.mkdir(parents=True, exist_ok=True)
19
+ target.write_text(json.dumps(value, indent=2) + "\n", encoding="utf-8")
20
+
21
+
22
+ def check_url(url: str) -> int:
23
+ request = Request(url, method="HEAD", headers={"User-Agent": "Maggie-SEO-Images/1.0"})
24
+ with urlopen(request, timeout=20) as response:
25
+ return response.status
26
+
27
+
28
+ def main() -> int:
29
+ parser = argparse.ArgumentParser()
30
+ sub = parser.add_subparsers(dest="command", required=True)
31
+ inventory = sub.add_parser("inventory")
32
+ inventory.add_argument("--project", default=".")
33
+ inventory.add_argument("--output", required=True)
34
+ plan = sub.add_parser("plan")
35
+ plan.add_argument("--inventory", required=True)
36
+ plan.add_argument("--widths", default="320,480,768,1080,1440,1920")
37
+ plan.add_argument("--provider", choices=("unsplash", "cloudinary"))
38
+ plan.add_argument("--confirm-external", action="store_true")
39
+ plan.add_argument("--output", required=True)
40
+ confirm = sub.add_parser("confirm")
41
+ confirm.add_argument("--manifest", required=True)
42
+ confirm.add_argument("--output", required=True)
43
+ validate = sub.add_parser("validate")
44
+ validate.add_argument("--manifest", required=True)
45
+ apply = sub.add_parser("apply")
46
+ apply.add_argument("--manifest", required=True)
47
+ apply.add_argument("--adapter", choices=("manifest",), required=True)
48
+ apply.add_argument("--output", required=True)
49
+ apply.add_argument("--confirm", action="store_true")
50
+ args = parser.parse_args()
51
+ try:
52
+ if args.command == "inventory":
53
+ result = inventory_project(args.project)
54
+ write_json(args.output, result)
55
+ print(json.dumps({"assets": len(result["assets"]), "output": str(Path(args.output).resolve())}, indent=2))
56
+ return 0
57
+ if args.command == "plan":
58
+ result = plan_variants(json.loads(Path(args.inventory).read_text(encoding="utf-8")), [int(item) for item in args.widths.split(",")], args.provider)
59
+ write_json(args.output, result)
60
+ print(json.dumps({"assets": len(result["assets"]), "output": str(Path(args.output).resolve())}, indent=2))
61
+ return 0
62
+ manifest = json.loads(Path(args.manifest).read_text(encoding="utf-8"))
63
+ if args.command == "confirm":
64
+ result = confirm_variants(manifest, check_url)
65
+ write_json(args.output, result)
66
+ print(json.dumps({"errors": validate_manifest(result), "output": str(Path(args.output).resolve())}, indent=2))
67
+ return 0 if not validate_manifest(result) else 1
68
+ if args.command == "validate":
69
+ errors = validate_manifest(manifest)
70
+ print(json.dumps({"status": "pass" if not errors else "fail", "errors": errors}, indent=2))
71
+ return 0 if not errors else 1
72
+ if not args.confirm:
73
+ raise ValueError("apply requires --confirm")
74
+ errors = validate_manifest(manifest)
75
+ if errors:
76
+ raise ValueError("manifest is invalid: " + "; ".join(errors))
77
+ applied = {**manifest, "apply": {"adapter": args.adapter, "status": "validated-local-manifest", "publicWrite": False}}
78
+ write_json(args.output, applied)
79
+ print(json.dumps({"status": applied["apply"]["status"], "publicWrite": False, "output": str(Path(args.output).resolve())}, indent=2))
80
+ return 0
81
+ except (OSError, ValueError, TypeError, json.JSONDecodeError) as exc:
82
+ parser.error(str(exc))
83
+ return 2
84
+
85
+
86
+ if __name__ == "__main__":
87
+ raise SystemExit(main())
@@ -0,0 +1,89 @@
1
+ #!/usr/bin/env python3
2
+ """Measure public pages with sampled PageSpeed evidence."""
3
+
4
+ from __future__ import annotations
5
+
6
+ import argparse
7
+ import json
8
+ import os
9
+ import sys
10
+ from pathlib import Path
11
+
12
+ sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "runtime"))
13
+ from maggie_pagespeed import STRATEGIES, fetch_pagespeed, load_json, make_report, validate_public_url # noqa: E402
14
+
15
+
16
+ def urls_from_args(args: argparse.Namespace) -> list[str]:
17
+ urls = list(args.url or [])
18
+ if args.url_file:
19
+ urls.extend(line.strip() for line in Path(args.url_file).read_text(encoding="utf-8").splitlines() if line.strip() and not line.lstrip().startswith("#"))
20
+ unique = list(dict.fromkeys(urls))
21
+ if not unique:
22
+ raise ValueError("at least one --url or --url-file is required")
23
+ for url in unique:
24
+ validate_public_url(url)
25
+ return unique
26
+
27
+
28
+ def main() -> int:
29
+ parser = argparse.ArgumentParser()
30
+ parser.add_argument("--url", action="append")
31
+ parser.add_argument("--url-file")
32
+ parser.add_argument("--strategy", default="mobile")
33
+ parser.add_argument("--samples", type=int, default=3)
34
+ parser.add_argument("--spacing-seconds", type=float, default=30)
35
+ parser.add_argument("--categories", default="performance")
36
+ parser.add_argument("--locale")
37
+ parser.add_argument("--fixture", help="JSON fixture containing an array of provider responses")
38
+ parser.add_argument("--baseline")
39
+ parser.add_argument("--output")
40
+ parser.add_argument("--fail-on", choices=("regression", "complete"), default=None)
41
+ parser.add_argument("--allow-low-sample", action="store_true")
42
+ parser.add_argument("--dry-run", action="store_true")
43
+ args = parser.parse_args()
44
+ try:
45
+ urls = urls_from_args(args)
46
+ strategies = [item.strip() for item in args.strategy.split(",") if item.strip()]
47
+ if not strategies or any(item not in STRATEGIES for item in strategies):
48
+ raise ValueError("strategy must contain only desktop and mobile")
49
+ if args.samples < 1 or (args.samples < 3 and not args.allow_low_sample):
50
+ raise ValueError("samples must be at least 3 unless --allow-low-sample is supplied")
51
+ if args.spacing_seconds < 0:
52
+ raise ValueError("spacing-seconds must not be negative")
53
+ planned_calls = len(urls) * len(strategies) * args.samples
54
+ if args.dry_run:
55
+ print(json.dumps({"urls": len(urls), "strategies": strategies, "samples": args.samples, "plannedProviderCalls": planned_calls, "spacingSeconds": args.spacing_seconds}, indent=2))
56
+ return 0
57
+ fixture_values = load_json(args.fixture) if args.fixture else None
58
+ fixture_values = fixture_values.get("responses", fixture_values) if fixture_values is not None else None
59
+ fixture_index = 0
60
+
61
+ def fetcher(url: str, strategy: str) -> dict:
62
+ nonlocal fixture_index
63
+ if fixture_values is not None:
64
+ if not isinstance(fixture_values, list) or not fixture_values:
65
+ raise ValueError("fixture must contain a non-empty response array")
66
+ response = fixture_values[min(fixture_index, len(fixture_values) - 1)]
67
+ fixture_index += 1
68
+ return response
69
+ return fetch_pagespeed(url, strategy, [item.strip() for item in args.categories.split(",") if item.strip()], args.locale, os.environ.get("PAGESPEED_API_KEY"))
70
+
71
+ baseline = load_json(args.baseline) if args.baseline else None
72
+ report = make_report(urls, strategies, args.samples, args.spacing_seconds, fetcher, baseline)
73
+ if args.output:
74
+ output = Path(args.output)
75
+ output.parent.mkdir(parents=True, exist_ok=True)
76
+ output.write_text(json.dumps(report, indent=2) + "\n", encoding="utf-8")
77
+ print(json.dumps(report, indent=2))
78
+ if args.fail_on == "regression" and report["summary"]["status"] == "fail":
79
+ return 2
80
+ if args.fail_on == "complete" and report["summary"]["status"] != "pass":
81
+ return 2
82
+ return 0
83
+ except (OSError, ValueError) as exc:
84
+ parser.error(str(exc))
85
+ return 2
86
+
87
+
88
+ if __name__ == "__main__":
89
+ raise SystemExit(main())
@@ -0,0 +1,108 @@
1
+ #!/usr/bin/env python3
2
+ """Plan and validate typed, conservative XML sitemaps."""
3
+
4
+ from __future__ import annotations
5
+
6
+ import argparse
7
+ import hashlib
8
+ import json
9
+ import sys
10
+ from pathlib import Path
11
+
12
+ sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "runtime"))
13
+ from maggie_sitemap import build_plan, parse_routes, validate_plan_data # noqa: E402
14
+
15
+
16
+ def write_json(path: str, value: dict) -> None:
17
+ target = Path(path)
18
+ target.parent.mkdir(parents=True, exist_ok=True)
19
+ target.write_text(json.dumps(value, indent=2) + "\n", encoding="utf-8")
20
+
21
+
22
+ def main() -> int:
23
+ parser = argparse.ArgumentParser()
24
+ sub = parser.add_subparsers(dest="command", required=True)
25
+ plan = sub.add_parser("plan")
26
+ plan.add_argument("--origin", required=True)
27
+ plan.add_argument("--routes-file", required=True)
28
+ plan.add_argument("--content-types", default="pages,posts,topics")
29
+ plan.add_argument("--chunk-target", type=int, default=500)
30
+ plan.add_argument("--previous-manifest")
31
+ plan.add_argument("--output", required=True)
32
+ validate = sub.add_parser("validate")
33
+ validate.add_argument("--plan", required=True)
34
+ apply = sub.add_parser("apply")
35
+ apply.add_argument("--plan", required=True)
36
+ apply.add_argument("--public-dir", required=True)
37
+ apply.add_argument("--backup-dir")
38
+ apply.add_argument("--confirm", action="store_true")
39
+ rollback = sub.add_parser("rollback")
40
+ rollback.add_argument("--backup-manifest", required=True)
41
+ rollback.add_argument("--public-dir", required=True)
42
+ rollback.add_argument("--confirm", action="store_true")
43
+ args = parser.parse_args()
44
+ try:
45
+ if args.command == "plan":
46
+ previous = json.loads(Path(args.previous_manifest).read_text(encoding="utf-8")) if args.previous_manifest else None
47
+ result = build_plan(parse_routes(Path(args.routes_file).read_text(encoding="utf-8")), args.origin, {item.strip() for item in args.content_types.split(",") if item.strip()}, args.chunk_target, previous)
48
+ write_json(args.output, result)
49
+ print(json.dumps({"status": result["validation"]["status"], "groups": len(result["groups"]), "redirects": len(result["redirects"]), "output": str(Path(args.output).resolve())}, indent=2))
50
+ return 0 if result["validation"]["status"] == "pass" else 1
51
+ if args.command == "rollback":
52
+ if not args.confirm:
53
+ raise ValueError("rollback requires --confirm")
54
+ backup = json.loads(Path(args.backup_manifest).read_text(encoding="utf-8"))
55
+ public = Path(args.public_dir)
56
+ restored = 0
57
+ for item in backup.get("files", []):
58
+ destination = public / item["filename"]
59
+ source = Path(args.backup_manifest).parent / item["filename"]
60
+ if item["existed"]:
61
+ destination.write_bytes(source.read_bytes())
62
+ restored += 1
63
+ elif destination.exists() and hashlib.sha256(destination.read_bytes()).hexdigest() == item["appliedSha256"]:
64
+ destination.unlink()
65
+ restored += 1
66
+ print(json.dumps({"status": "rolled-back", "restored": restored}, indent=2))
67
+ return 0
68
+ plan_value = json.loads(Path(args.plan).read_text(encoding="utf-8"))
69
+ if args.command == "validate":
70
+ result = validate_plan_data(plan_value)
71
+ print(json.dumps(result, indent=2))
72
+ return 0 if result["status"] == "pass" else 1
73
+ if not args.confirm:
74
+ raise ValueError("apply requires --confirm")
75
+ validation = validate_plan_data(plan_value)
76
+ if validation["status"] != "pass":
77
+ raise ValueError("plan is invalid: " + "; ".join(validation["errors"]))
78
+ target = Path(args.public_dir)
79
+ target.mkdir(parents=True, exist_ok=True)
80
+ backup_root = Path(args.backup_dir) if args.backup_dir else target / ".maggie-sitemap-backups" / plan_value["planId"]
81
+ backup_root.mkdir(parents=True, exist_ok=True)
82
+ backup_files = []
83
+ output_files = [*(chunk for group in plan_value.get("groups", []) for chunk in group.get("chunks", [])), {"filename": "sitemap.xml", "xml": plan_value["index"]["xml"]}]
84
+ for item in output_files:
85
+ destination = target / item["filename"]
86
+ existed = destination.exists()
87
+ if existed:
88
+ (backup_root / item["filename"]).parent.mkdir(parents=True, exist_ok=True)
89
+ (backup_root / item["filename"]).write_bytes(destination.read_bytes())
90
+ content = item["xml"].encode("utf-8")
91
+ backup_files.append({"filename": item["filename"], "existed": existed, "appliedSha256": hashlib.sha256(content).hexdigest()})
92
+ for group in plan_value.get("groups", []):
93
+ for chunk in group.get("chunks", []):
94
+ (target / chunk["filename"]).write_text(chunk["xml"], encoding="utf-8")
95
+ (target / "sitemap.xml").write_text(plan_value["index"]["xml"], encoding="utf-8")
96
+ if plan_value.get("redirects"):
97
+ write_json(str(target / "sitemap-redirect-plan.json"), {"redirects": plan_value["redirects"]})
98
+ backup_manifest = {"schemaVersion": "maggie-seo-sitemap-rollback.v1", "planId": plan_value["planId"], "publicDir": str(target.resolve()), "files": backup_files}
99
+ write_json(str(backup_root / "backup-manifest.json"), backup_manifest)
100
+ print(json.dumps({"status": "applied-local-public-dir", "publicWrite": True, "files": len(output_files), "backupManifest": str((backup_root / "backup-manifest.json").resolve())}, indent=2))
101
+ return 0
102
+ except (OSError, ValueError, TypeError, json.JSONDecodeError) as exc:
103
+ parser.error(str(exc))
104
+ return 2
105
+
106
+
107
+ if __name__ == "__main__":
108
+ raise SystemExit(main())
@@ -134,10 +134,11 @@ def hreflang_check(url: str, page: PageParser, expected: set[str]) -> dict:
134
134
  return {"ok": not missing and not reciprocal_errors, "missing": missing, "reciprocal_errors": reciprocal_errors, "links": links}
135
135
 
136
136
 
137
- def sitemap_urls(base: str) -> list[str]:
138
- """Resolve a sitemap index and return its public page URLs."""
137
+ def sitemap_urls(base: str) -> tuple[list[str], list[dict[str, str]]]:
138
+ """Resolve a sitemap index and return page URLs plus raw loc violations."""
139
139
  pending = [urljoin(base + "/", "sitemap.xml")]
140
140
  pages = []
141
+ violations = []
141
142
  seen = set()
142
143
  while pending:
143
144
  sitemap_url = pending.pop(0)
@@ -145,12 +146,16 @@ def sitemap_urls(base: str) -> list[str]:
145
146
  continue
146
147
  seen.add(sitemap_url)
147
148
  _, _, body = fetch(sitemap_url)
148
- locs = [urljoin(base + "/", value.strip()) for value in re.findall(r"<loc>\s*(.*?)\s*</loc>", body, re.I | re.S)]
149
+ raw_locs = [value.strip() for value in re.findall(r"<loc>\s*(.*?)\s*</loc>", body, re.I | re.S)]
150
+ for value in raw_locs:
151
+ if not re.match(r"^https?://[^\s]+$", value, re.I):
152
+ violations.append({"sitemap": sitemap_url, "loc": value, "reason": "sitemap loc must be an absolute HTTP(S) URL"})
153
+ locs = [urljoin(base + "/", value) for value in raw_locs]
149
154
  if re.search(r"<sitemapindex\b", body, re.I):
150
155
  pending.extend(locs)
151
156
  else:
152
157
  pages.extend(locs)
153
- return list(dict.fromkeys(pages))
158
+ return list(dict.fromkeys(pages)), violations
154
159
 
155
160
 
156
161
  def main() -> int:
@@ -215,7 +220,8 @@ def main() -> int:
215
220
  crawl = {"enabled": args.crawl, "passed": True, "pages": []}
216
221
  if args.crawl:
217
222
  try:
218
- urls = sitemap_urls(base)[: args.max_pages]
223
+ urls, sitemap_violations = sitemap_urls(base)
224
+ urls = urls[: args.max_pages]
219
225
  for page_url in urls:
220
226
  try:
221
227
  status, content_type, body = fetch(page_url)
@@ -232,7 +238,8 @@ def main() -> int:
232
238
  except Exception as exc:
233
239
  page_checks = {"url": page_url, "passed": False, "error": type(exc).__name__}
234
240
  crawl["pages"].append(page_checks)
235
- crawl["passed"] = bool(urls) and len(crawl["pages"]) == len(urls) and all(item["passed"] for item in crawl["pages"])
241
+ crawl["sitemap_loc_violations"] = sitemap_violations
242
+ crawl["passed"] = not sitemap_violations and bool(urls) and len(crawl["pages"]) == len(urls) and all(item["passed"] for item in crawl["pages"])
236
243
  crawl["url_count"] = len(urls)
237
244
  except Exception as exc:
238
245
  crawl = {"enabled": True, "passed": False, "error": type(exc).__name__, "pages": []}
@@ -0,0 +1,144 @@
1
+ """Safe image inventory, responsive variant planning, and confirmation helpers."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import hashlib
6
+ import mimetypes
7
+ import re
8
+ from pathlib import Path
9
+ from typing import Any, Callable
10
+ from urllib.parse import parse_qsl, urlencode, urlparse, urlunparse
11
+
12
+
13
+ IMAGE_TYPES = {".avif", ".gif", ".jpeg", ".jpg", ".png", ".webp"}
14
+ SKIP_DIRS = {".git", ".maggie", "node_modules", "dist", "build", ".astro"}
15
+ THIRD_PARTY_PROVIDERS = {"unsplash", "cloudinary"}
16
+
17
+
18
+ def asset_id(value: str) -> str:
19
+ return "asset-" + hashlib.sha256(value.encode("utf-8")).hexdigest()[:16]
20
+
21
+
22
+ def image_dimensions(path: Path) -> tuple[int | None, int | None]:
23
+ try:
24
+ data = path.read_bytes()
25
+ except OSError:
26
+ return None, None
27
+ if data.startswith(b"\x89PNG\r\n\x1a\n") and len(data) >= 24:
28
+ return int.from_bytes(data[16:20], "big"), int.from_bytes(data[20:24], "big")
29
+ if data.startswith(b"RIFF") and data[8:12] == b"WEBP" and len(data) >= 30:
30
+ if data[12:16] == b"VP8X":
31
+ return 1 + int.from_bytes(data[24:27], "little"), 1 + int.from_bytes(data[27:30], "little")
32
+ if data.startswith(b"\xff\xd8"):
33
+ index = 2
34
+ while index + 9 < len(data):
35
+ if data[index] != 0xFF:
36
+ index += 1
37
+ continue
38
+ marker = data[index + 1]
39
+ index += 2
40
+ if marker in {0xD8, 0xD9}:
41
+ continue
42
+ length = int.from_bytes(data[index:index + 2], "big")
43
+ if marker in set(range(0xC0, 0xC4)) | set(range(0xC5, 0xC8)) | set(range(0xC9, 0xCC)) | set(range(0xCD, 0xD0)):
44
+ return int.from_bytes(data[index + 5:index + 7], "big"), int.from_bytes(data[index + 3:index + 5], "big")
45
+ index += length
46
+ return None, None
47
+
48
+
49
+ def inventory_project(project: str | Path) -> dict[str, Any]:
50
+ root = Path(project).resolve()
51
+ assets = []
52
+ for path in sorted(root.rglob("*")):
53
+ if not path.is_file() or path.suffix.lower() not in IMAGE_TYPES or any(part in SKIP_DIRS for part in path.parts):
54
+ continue
55
+ relative = path.relative_to(root).as_posix()
56
+ width, height = image_dimensions(path)
57
+ assets.append({"assetId": asset_id(relative), "source": {"path": relative, "visibility": "public" if relative.startswith("public/") else "unknown"}, "mimeType": mimetypes.guess_type(path.name)[0] or "application/octet-stream", "bytes": path.stat().st_size, "intrinsic": {"width": width, "height": height}, "altText": {"status": "unknown"}})
58
+ return {"schemaVersion": "maggie-seo-image-inventory.v1", "assets": assets, "provenance": {"project": str(root), "redactions": ["absolute_project_path_in_public_reports"]}}
59
+
60
+
61
+ def transform_third_party_url(url: str, provider: str, width: int) -> str:
62
+ if provider not in THIRD_PARTY_PROVIDERS:
63
+ raise ValueError("third-party provider is not allowlisted")
64
+ parsed = urlparse(url)
65
+ if parsed.scheme not in {"http", "https"} or not parsed.hostname:
66
+ raise ValueError("third-party source must be an absolute HTTP(S) URL")
67
+ if parsed.username or parsed.password or parsed.fragment:
68
+ raise ValueError("third-party source contains unsafe URL data")
69
+ if width < 1:
70
+ raise ValueError("variant width must be positive")
71
+ if provider == "unsplash":
72
+ query = dict(parse_qsl(parsed.query, keep_blank_values=True))
73
+ query.update({"w": str(width), "auto": "format"})
74
+ return urlunparse((parsed.scheme, parsed.netloc, parsed.path, parsed.params, urlencode(query), ""))
75
+ if "/upload/" not in parsed.path:
76
+ raise ValueError("cloudinary URL must contain an /upload/ segment")
77
+ path = parsed.path.replace("/upload/", f"/upload/w_{width},f_auto/", 1)
78
+ return urlunparse((parsed.scheme, parsed.netloc, path, parsed.params, parsed.query, ""))
79
+
80
+
81
+ def plan_variants(inventory: dict[str, Any], widths: list[int], provider: str | None = None) -> dict[str, Any]:
82
+ if not widths or any(width < 1 for width in widths):
83
+ raise ValueError("widths must contain positive integers")
84
+ assets = []
85
+ for item in inventory.get("assets", []):
86
+ source = item.get("source", {})
87
+ source_url = source.get("url")
88
+ source_path = source.get("path")
89
+ variants = []
90
+ for width in sorted(set(widths)):
91
+ if source_url and provider:
92
+ url = transform_third_party_url(source_url, provider, width)
93
+ status = "planned"
94
+ elif source_path:
95
+ path = Path(source_path)
96
+ url = str(path.with_name(f"{path.stem}-{width}{path.suffix}"))
97
+ # Planning never confirms a derived file merely because the
98
+ # original has the same intrinsic width. Confirmation requires
99
+ # checking the candidate artifact itself.
100
+ status = "planned"
101
+ else:
102
+ continue
103
+ variants.append({"width": width, "format": Path(urlparse(url).path).suffix.lstrip(".") or "original", "url": url, "status": status, "evidence": {}})
104
+ assets.append({**item, "variants": variants, "markupPolicy": {"allowedWidths": [item["width"] for item in variants if item["status"] == "confirmed"], "missingVariantPolicy": "omit", "fetchPriority": "auto"}})
105
+ return {"schemaVersion": "maggie-seo-image-variants.v1", "manifestId": "images-" + hashlib.sha256(str(len(assets)).encode()).hexdigest()[:12], "createdAt": __import__("datetime").datetime.now(__import__("datetime").timezone.utc).isoformat().replace("+00:00", "Z"), "assets": assets, "provenance": {"redactions": ["private_paths", "credentials"]}}
106
+
107
+
108
+ def confirm_variants(manifest: dict[str, Any], checker: Callable[[str], int]) -> dict[str, Any]:
109
+ result = {**manifest, "assets": []}
110
+ for asset in manifest.get("assets", []):
111
+ updated = {**asset, "variants": []}
112
+ for variant in asset.get("variants", []):
113
+ candidate = {**variant}
114
+ if candidate.get("status") == "planned":
115
+ try:
116
+ status = checker(candidate["url"])
117
+ candidate["status"] = "confirmed" if status == 200 else "failed"
118
+ candidate["evidence"] = {"status": status}
119
+ except Exception as exc:
120
+ candidate["status"] = "failed"
121
+ candidate["evidence"] = {"error": type(exc).__name__}
122
+ updated["variants"].append(candidate)
123
+ updated["markupPolicy"] = {**asset.get("markupPolicy", {}), "allowedWidths": [v["width"] for v in updated["variants"] if v.get("status") == "confirmed"]}
124
+ result["assets"].append(updated)
125
+ return result
126
+
127
+
128
+ def confirmed_srcset(asset: dict[str, Any]) -> str:
129
+ return ", ".join(f"{variant['url']} {variant['width']}w" for variant in asset.get("variants", []) if variant.get("status") == "confirmed")
130
+
131
+
132
+ def validate_manifest(manifest: dict[str, Any]) -> list[str]:
133
+ errors = []
134
+ for asset in manifest.get("assets", []):
135
+ for variant in asset.get("variants", []):
136
+ if variant.get("status") == "confirmed" and not variant.get("evidence"):
137
+ errors.append(f"confirmed variant lacks evidence: {asset.get('assetId')}")
138
+ if variant.get("status") == "confirmed" and not variant.get("url"):
139
+ errors.append(f"confirmed variant lacks URL: {asset.get('assetId')}")
140
+ allowed = set(asset.get("markupPolicy", {}).get("allowedWidths", []))
141
+ actual = {variant.get("width") for variant in asset.get("variants", []) if variant.get("status") == "confirmed"}
142
+ if allowed != actual:
143
+ errors.append(f"markup allowedWidths does not match confirmed variants: {asset.get('assetId')}")
144
+ return errors
@@ -0,0 +1,213 @@
1
+ """Provider-neutral PageSpeed sampling, normalization, and comparison helpers."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ import os
7
+ import re
8
+ import statistics
9
+ import time
10
+ from datetime import datetime, timezone
11
+ from pathlib import Path
12
+ from typing import Any, Callable
13
+ from urllib.parse import parse_qsl, urlencode, urlparse, urlunparse
14
+ from urllib.request import Request, urlopen
15
+
16
+
17
+ METRICS = ("performanceScore", "lcp", "cls", "inp", "tbt", "fcp", "ttfb")
18
+ STRATEGIES = ("desktop", "mobile")
19
+ DEFAULT_THRESHOLDS = {
20
+ "performanceScore": {"direction": "decrease", "absolute": 0.05},
21
+ "lcp": {"direction": "increase", "relative": 0.10},
22
+ "cls": {"direction": "increase", "absolute": 0.05},
23
+ "inp": {"direction": "increase", "relative": 0.10},
24
+ "tbt": {"direction": "increase", "relative": 0.10},
25
+ "fcp": {"direction": "increase", "relative": 0.10},
26
+ "ttfb": {"direction": "increase", "relative": 0.10},
27
+ }
28
+
29
+
30
+ class PageSpeedError(RuntimeError):
31
+ """A safe, typed provider failure without credentials or response bodies."""
32
+
33
+ def __init__(self, kind: str, message: str) -> None:
34
+ self.kind = kind
35
+ super().__init__(message)
36
+
37
+
38
+ def now_iso() -> str:
39
+ return datetime.now(timezone.utc).isoformat().replace("+00:00", "Z")
40
+
41
+
42
+ def validate_public_url(value: str) -> str:
43
+ parsed = urlparse(value.strip())
44
+ if parsed.scheme not in {"http", "https"} or not parsed.hostname:
45
+ raise ValueError("URL must be an absolute HTTP(S) URL")
46
+ if parsed.username or parsed.password:
47
+ raise ValueError("URL must not contain credentials")
48
+ if parsed.fragment:
49
+ raise ValueError("URL must not contain a fragment")
50
+ return value.strip()
51
+
52
+
53
+ def redact_url(value: str) -> str:
54
+ parsed = urlparse(value)
55
+ safe_query = [(key, "[redacted]") if re.search(r"token|key|secret|password", key, re.I) else (key, val) for key, val in parse_qsl(parsed.query, keep_blank_values=True)]
56
+ return urlunparse((parsed.scheme, parsed.netloc, parsed.path, parsed.params, urlencode(safe_query), ""))
57
+
58
+
59
+ def _number(value: Any) -> float | None:
60
+ return float(value) if isinstance(value, (int, float)) and not isinstance(value, bool) else None
61
+
62
+
63
+ def _audit_number(audits: dict[str, Any], *names: str) -> float | None:
64
+ for name in names:
65
+ value = audits.get(name, {})
66
+ if isinstance(value, dict):
67
+ number = _number(value.get("numericValue"))
68
+ if number is not None:
69
+ return number
70
+ return None
71
+
72
+
73
+ def _field_metric(metrics: dict[str, Any], *names: str) -> dict[str, Any]:
74
+ for name in names:
75
+ value = metrics.get(name)
76
+ if isinstance(value, dict):
77
+ percentile = _number(value.get("percentile"))
78
+ return {"value": percentile, "category": value.get("category"), "source": "field"}
79
+ return {"value": None, "source": "field", "status": "not_available"}
80
+
81
+
82
+ def normalize_response(response: dict[str, Any], requested_url: str, strategy: str, sampled_at: str | None = None) -> dict[str, Any]:
83
+ lighthouse = response.get("lighthouseResult") or {}
84
+ audits = lighthouse.get("audits") if isinstance(lighthouse.get("audits"), dict) else {}
85
+ categories = lighthouse.get("categories") if isinstance(lighthouse.get("categories"), dict) else {}
86
+ performance = categories.get("performance") if isinstance(categories.get("performance"), dict) else {}
87
+ loading = response.get("loadingExperience") if isinstance(response.get("loadingExperience"), dict) else {}
88
+ field_metrics = loading.get("metrics") if isinstance(loading.get("metrics"), dict) else {}
89
+ lab = {
90
+ "performanceScore": _number(performance.get("score")),
91
+ "lcp": _audit_number(audits, "largest-contentful-paint"),
92
+ "cls": _audit_number(audits, "cumulative-layout-shift"),
93
+ "inp": _audit_number(audits, "interaction-to-next-paint", "experimental-interaction-to-next-paint"),
94
+ "tbt": _audit_number(audits, "total-blocking-time"),
95
+ "fcp": _audit_number(audits, "first-contentful-paint"),
96
+ "ttfb": _audit_number(audits, "server-response-time", "time-to-first-byte"),
97
+ }
98
+ field = {
99
+ "lcp": _field_metric(field_metrics, "LARGEST_CONTENTFUL_PAINT_MS"),
100
+ "cls": _field_metric(field_metrics, "CUMULATIVE_LAYOUT_SHIFT_SCORE"),
101
+ "inp": _field_metric(field_metrics, "INTERACTION_TO_NEXT_PAINT"),
102
+ }
103
+ return {
104
+ "requestedUrl": redact_url(requested_url),
105
+ "finalUrl": redact_url(str(lighthouse.get("finalUrl") or response.get("id") or requested_url)),
106
+ "strategy": strategy,
107
+ "sampledAt": sampled_at or now_iso(),
108
+ "fetchTime": lighthouse.get("fetchTime"),
109
+ "duplicateFetchTime": False,
110
+ "status": "ok",
111
+ "providerVersion": lighthouse.get("lighthouseVersion"),
112
+ "warnings": lighthouse.get("runWarnings") if isinstance(lighthouse.get("runWarnings"), list) else [],
113
+ "lab": lab,
114
+ "fieldData": {
115
+ "status": "available" if field_metrics else "not_available",
116
+ "source": "pagespeed-loading-experience",
117
+ "metrics": field,
118
+ },
119
+ }
120
+
121
+
122
+ def fetch_pagespeed(url: str, strategy: str, categories: list[str] | None = None, locale: str | None = None, api_key: str | None = None, timeout: int = 45) -> dict[str, Any]:
123
+ validate_public_url(url)
124
+ if strategy not in STRATEGIES:
125
+ raise ValueError("strategy must be desktop or mobile")
126
+ query = {"url": url, "strategy": strategy}
127
+ for category in categories or ["performance"]:
128
+ query.setdefault("category", category)
129
+ if locale:
130
+ query["locale"] = locale
131
+ if api_key:
132
+ query["key"] = api_key
133
+ endpoint = "https://pagespeedonline.googleapis.com/pagespeedonline/v5/runPagespeed?" + urlencode(query)
134
+ request = Request(endpoint, headers={"User-Agent": "Maggie-SEO-Performance/1.0", "Accept": "application/json"})
135
+ try:
136
+ with urlopen(request, timeout=timeout) as response:
137
+ if response.status != 200:
138
+ raise PageSpeedError(f"http_{response.status}", f"PageSpeed returned HTTP {response.status}")
139
+ return json.loads(response.read(5_000_000).decode("utf-8"))
140
+ except PageSpeedError:
141
+ raise
142
+ except TimeoutError as exc:
143
+ raise PageSpeedError("timeout", "PageSpeed request timed out") from exc
144
+ except json.JSONDecodeError as exc:
145
+ raise PageSpeedError("invalid_json", "PageSpeed returned invalid JSON") from exc
146
+ except Exception as exc:
147
+ status = getattr(exc, "code", None)
148
+ if status:
149
+ raise PageSpeedError(f"http_{status}", f"PageSpeed returned HTTP {status}") from exc
150
+ raise PageSpeedError("network", f"PageSpeed request failed: {type(exc).__name__}") from exc
151
+
152
+
153
+ def aggregate_samples(samples: list[dict[str, Any]], minimum: int = 3) -> dict[str, Any]:
154
+ independent = [sample for sample in samples if sample.get("status") == "ok" and not sample.get("duplicateFetchTime")]
155
+ aggregate: dict[str, Any] = {"independentSampleCount": len(independent), "status": "pass" if len(independent) >= minimum else "inconclusive", "metrics": {}}
156
+ for metric in METRICS:
157
+ values = [_number((sample.get("lab") or {}).get(metric)) for sample in independent]
158
+ values = [value for value in values if value is not None]
159
+ aggregate["metrics"][metric] = {"median": statistics.median(values) if values else None, "min": min(values) if values else None, "max": max(values) if values else None, "spread": max(values) - min(values) if values else None, "sampleCount": len(values), "status": "available" if values else "not_available"}
160
+ return aggregate
161
+
162
+
163
+ def compare_aggregate(current: dict[str, Any], baseline: dict[str, Any] | None, thresholds: dict[str, dict[str, float]] | None = None) -> dict[str, Any]:
164
+ if not baseline:
165
+ return {"status": "no_baseline", "baselineId": None, "changes": [], "reasons": ["no baseline supplied"]}
166
+ if current.get("independentSampleCount", 0) < 3 or baseline.get("independentSampleCount", 0) < 3:
167
+ return {"status": "inconclusive", "baselineId": baseline.get("reportId"), "changes": [], "reasons": ["both current and baseline require at least three independent samples"]}
168
+ policy = thresholds or DEFAULT_THRESHOLDS
169
+ changes = []
170
+ for metric, rule in policy.items():
171
+ before = _number((baseline.get("metrics", {}).get(metric) or {}).get("median"))
172
+ after = _number((current.get("metrics", {}).get(metric) or {}).get("median"))
173
+ if before is None or after is None:
174
+ continue
175
+ delta = after - before
176
+ breach = delta <= -rule.get("absolute", float("inf")) if rule["direction"] == "decrease" and "absolute" in rule else delta >= rule.get("absolute", float("inf")) if rule["direction"] == "increase" and "absolute" in rule else (after <= before * (1 - rule.get("relative", 0)) if rule["direction"] == "decrease" else after >= before * (1 + rule.get("relative", 0)))
177
+ changes.append({"metric": metric, "before": before, "after": after, "delta": delta, "breach": bool(breach), "rule": rule})
178
+ breached = [change for change in changes if change["breach"]]
179
+ return {"status": "regression" if len(breached) >= 1 else "no_regression", "baselineId": baseline.get("reportId"), "changes": changes, "reasons": [f"{item['metric']} exceeded configured threshold" for item in breached]}
180
+
181
+
182
+ def make_report(urls: list[str], strategies: list[str], samples: int, spacing_seconds: float, fetcher: Callable[[str, str], dict[str, Any]], baseline: dict[str, Any] | None = None, minimum: int = 3) -> dict[str, Any]:
183
+ pages = []
184
+ for url in urls:
185
+ validate_public_url(url)
186
+ for strategy in strategies:
187
+ records = []
188
+ seen_fetch_times: set[str] = set()
189
+ for index in range(samples):
190
+ if index and spacing_seconds:
191
+ time.sleep(spacing_seconds)
192
+ try:
193
+ record = normalize_response(fetcher(url, strategy), url, strategy)
194
+ fetch_time = record.get("fetchTime")
195
+ if fetch_time and fetch_time in seen_fetch_times:
196
+ record["duplicateFetchTime"] = True
197
+ elif fetch_time:
198
+ seen_fetch_times.add(fetch_time)
199
+ except PageSpeedError as exc:
200
+ record = {"requestedUrl": redact_url(url), "strategy": strategy, "sampledAt": now_iso(), "status": "error", "error": {"kind": exc.kind, "message": str(exc)}}
201
+ records.append(record)
202
+ aggregate = aggregate_samples(records, minimum)
203
+ baseline_page = None
204
+ if baseline:
205
+ baseline_page = next((item.get("aggregate") for item in baseline.get("pages", []) if item.get("requestedUrl") == redact_url(url) and item.get("strategy") == strategy), None)
206
+ pages.append({"requestedUrl": redact_url(url), "strategy": strategy, "samples": records, "aggregate": aggregate, "fieldData": {"status": "available" if any((item.get("fieldData") or {}).get("status") == "available" for item in records) else "not_available"}, "comparison": compare_aggregate(aggregate, baseline_page)})
207
+ statuses = [page["comparison"]["status"] for page in pages]
208
+ summary = "fail" if "regression" in statuses else "inconclusive" if any(status == "inconclusive" or page["aggregate"]["status"] == "inconclusive" for status, page in zip(statuses, pages)) else "pass"
209
+ return {"schemaVersion": "maggie-seo-performance.v1", "reportId": "perf-" + datetime.now(timezone.utc).strftime("%Y%m%d%H%M%S"), "createdAt": now_iso(), "provider": "pagespeed-insights", "providerApiVersion": "v5", "policy": {"strategies": strategies, "requestedSamples": samples, "minimumIndependentSamples": minimum, "spacingSeconds": spacing_seconds, "duplicateFetchTimePolicy": "exclude"}, "pages": pages, "summary": {"status": summary}, "provenance": {"commands": [], "redactions": ["api_key", "credentialed_urls"]}}
210
+
211
+
212
+ def load_json(path: str | Path) -> dict[str, Any]:
213
+ return json.loads(Path(path).read_text(encoding="utf-8"))
@@ -0,0 +1,131 @@
1
+ """Typed sitemap planning, chunking, validation, and redirect planning."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import hashlib
6
+ import re
7
+ from datetime import datetime, timezone
8
+ from html import escape
9
+ from urllib.parse import parse_qs, urlparse
10
+ from xml.etree import ElementTree
11
+
12
+ HARD_URL_LIMIT = 50_000
13
+ HARD_BYTES_LIMIT = 52_428_800
14
+ DEFAULT_CHUNK_TARGET = 500
15
+ EXCLUDED_PATHS = re.compile(r"/(search|find|login|draft|preview)(/|$)", re.I)
16
+
17
+
18
+ def absolute_url(url: str, origin: str) -> bool:
19
+ parsed = urlparse(url)
20
+ return parsed.scheme in {"http", "https"} and bool(parsed.netloc) and not parsed.fragment and parsed.netloc == urlparse(origin).netloc
21
+
22
+
23
+ def parse_routes(text: str) -> list[dict[str, str]]:
24
+ routes = []
25
+ for line in text.splitlines():
26
+ if not line.strip() or line.lstrip().startswith("#"):
27
+ continue
28
+ parts = line.split("\t")
29
+ if len(parts) < 2:
30
+ raise ValueError("route lines must be type<TAB>absolute-url<TAB>optional-lastmod")
31
+ item = {"type": parts[0].strip(), "url": parts[1].strip()}
32
+ if len(parts) > 2 and parts[2].strip():
33
+ item["lastmod"] = parts[2].strip()
34
+ routes.append(item)
35
+ return routes
36
+
37
+
38
+ def filter_routes(routes: list[dict[str, str]], origin: str, content_types: set[str]) -> tuple[dict[str, list[dict[str, str]]], list[dict[str, str]]]:
39
+ groups = {item: [] for item in sorted(content_types)}
40
+ excluded = []
41
+ for route in routes:
42
+ url = route.get("url", "")
43
+ parsed = urlparse(url)
44
+ reason = None
45
+ if route.get("type") not in content_types:
46
+ reason = "content type not enabled"
47
+ elif not absolute_url(url, origin):
48
+ reason = "URL must be an absolute HTTP(S) URL on the approved origin"
49
+ elif EXCLUDED_PATHS.search(parsed.path) or any(key in parse_qs(parsed.query) for key in ("q", "search", "filter", "page")):
50
+ reason = "search, filter, or non-canonical route"
51
+ elif route.get("type", "").lower() in {"rss", "atom", "feed"}:
52
+ reason = "feed is not an HTML sitemap entry"
53
+ if reason:
54
+ excluded.append({"url": url, "type": route.get("type", ""), "reason": reason})
55
+ else:
56
+ groups.setdefault(route["type"], []).append(route)
57
+ for key in groups:
58
+ groups[key] = list({item["url"]: item for item in groups[key]}.values())
59
+ return groups, excluded
60
+
61
+
62
+ def xml_file(urls: list[dict[str, str]]) -> str:
63
+ rows = []
64
+ for item in urls:
65
+ lastmod = f"<lastmod>{escape(item['lastmod'])}</lastmod>" if item.get("lastmod") else ""
66
+ rows.append(f"<url><loc>{escape(item['url'])}</loc>{lastmod}</url>")
67
+ return '<?xml version="1.0" encoding="UTF-8"?>\n<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">' + "".join(rows) + "</urlset>\n"
68
+
69
+
70
+ def build_plan(routes: list[dict[str, str]], origin: str, content_types: set[str], chunk_target: int = DEFAULT_CHUNK_TARGET, previous: dict | None = None) -> dict:
71
+ if chunk_target < 1 or chunk_target > HARD_URL_LIMIT:
72
+ raise ValueError("chunk target must be between 1 and 50000")
73
+ groups, excluded = filter_routes(routes, origin, content_types)
74
+ group_plans = []
75
+ sitemap_urls = []
76
+ for content_type in sorted(content_types):
77
+ chunks = []
78
+ entries = groups.get(content_type, []) or []
79
+ for index in range(0, max(len(entries), 1), chunk_target):
80
+ chunk_entries = entries[index:index + chunk_target]
81
+ filename = f"sitemap-{content_type}-{index // chunk_target + 1}.xml"
82
+ xml = xml_file(chunk_entries)
83
+ url = origin.rstrip("/") + "/" + filename
84
+ chunks.append({"filename": filename, "url": url, "entries": len(chunk_entries), "bytes": len(xml.encode()), "sha256": hashlib.sha256(xml.encode()).hexdigest(), "xml": xml})
85
+ sitemap_urls.append(url)
86
+ group_plans.append({"contentType": content_type, "chunks": chunks})
87
+ index_xml = '<?xml version="1.0" encoding="UTF-8"?>\n<sitemapindex xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">' + "".join(f"<sitemap><loc>{escape(url)}</loc></sitemap>" for url in sitemap_urls) + "</sitemapindex>\n"
88
+ current = set(sitemap_urls)
89
+ old = set((previous or {}).get("sitemapUrls", []))
90
+ redirects = [{"from": url, "to": origin.rstrip("/") + "/sitemap.xml", "reason": "previously advertised sitemap removed"} for url in sorted(old - current)]
91
+ plan = {"schemaVersion": "maggie-seo-sitemap-plan.v1", "planId": "sitemap-" + datetime.now(timezone.utc).strftime("%Y%m%d%H%M%S"), "origin": origin.rstrip("/"), "groups": group_plans, "index": {"url": origin.rstrip("/") + "/sitemap.xml", "bytes": len(index_xml.encode()), "xml": index_xml, "sitemapUrls": sitemap_urls}, "redirects": redirects, "excluded": excluded}
92
+ plan["validation"] = validate_plan_data(plan)
93
+ return plan
94
+
95
+
96
+ def validate_plan_data(plan: dict) -> dict:
97
+ errors = []
98
+ warnings = []
99
+ origin = plan.get("origin", "")
100
+ if not urlparse(origin).scheme or not urlparse(origin).netloc:
101
+ errors.append("origin must be an absolute HTTP(S) URL")
102
+ expected_urls = []
103
+ for group in plan.get("groups", []):
104
+ for chunk in group.get("chunks", []):
105
+ expected_urls.append(chunk.get("url"))
106
+ if chunk.get("entries", 0) > HARD_URL_LIMIT:
107
+ errors.append(f"chunk exceeds URL limit: {chunk.get('filename')}")
108
+ if chunk.get("bytes", 0) > HARD_BYTES_LIMIT:
109
+ errors.append(f"chunk exceeds byte limit: {chunk.get('filename')}")
110
+ try:
111
+ root = ElementTree.fromstring(chunk.get("xml", ""))
112
+ locs = [element.text or "" for element in root.iter() if element.tag.endswith("loc")]
113
+ if any(not absolute_url(value, origin) for value in locs):
114
+ errors.append(f"chunk contains a relative or off-origin loc: {chunk.get('filename')}")
115
+ except ElementTree.ParseError:
116
+ errors.append(f"chunk is not valid XML: {chunk.get('filename')}")
117
+ index = plan.get("index", {})
118
+ if index.get("bytes", 0) > HARD_BYTES_LIMIT:
119
+ errors.append("sitemap index exceeds byte limit")
120
+ try:
121
+ index_root = ElementTree.fromstring(index.get("xml", ""))
122
+ index_locs = [element.text or "" for element in index_root.iter() if element.tag.endswith("loc")]
123
+ if any(not absolute_url(value, origin) for value in index_locs):
124
+ errors.append("sitemap index contains a relative or off-origin loc")
125
+ if set(index_locs) != set(expected_urls):
126
+ errors.append("sitemap index references do not match generated chunks")
127
+ except ElementTree.ParseError:
128
+ errors.append("sitemap index is not valid XML")
129
+ if not expected_urls:
130
+ warnings.append("no sitemap chunks generated")
131
+ return {"status": "pass" if not errors else "fail", "errors": errors, "warnings": warnings}
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@topy-ai/maggie",
3
- "version": "0.5.0",
3
+ "version": "0.6.0",
4
4
  "description": "Install and manage Maggie Skills for AI coding agents",
5
5
  "license": "MIT",
6
6
  "type": "module",