@topy-ai/maggie 0.5.1 → 0.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +33 -3
- package/bin/maggie.js +9 -0
- package/bundled-contracts/maggie-seo/image-variants-v1.schema.json +14 -0
- package/bundled-contracts/maggie-seo/performance-report-v1.schema.json +46 -0
- package/bundled-contracts/maggie-seo/sitemap-plan-v1.schema.json +16 -0
- package/bundled-skills/maggie-seo-geo/SKILL.md +33 -0
- package/bundled-tools/clis/maggie_images.py +87 -0
- package/bundled-tools/clis/maggie_performance.py +89 -0
- package/bundled-tools/clis/maggie_sitemap.py +108 -0
- package/bundled-tools/runtime/maggie_media.py +144 -0
- package/bundled-tools/runtime/maggie_pagespeed.py +213 -0
- package/bundled-tools/runtime/maggie_sitemap.py +131 -0
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -46,7 +46,7 @@ Then invoke the installed skills from your coding agent, for example:
|
|
|
46
46
|
|
|
47
47
|
Maggie keeps the existing project foundation and asks for decisions before
|
|
48
48
|
shared routes, analytics, or publishing boundaries change. The current
|
|
49
|
-
package ships
|
|
49
|
+
package ships 18 installable skills and a local-first MaggieDash foundation.
|
|
50
50
|
|
|
51
51
|
## Command surface
|
|
52
52
|
|
|
@@ -66,6 +66,9 @@ maggie memory ... # confirmed preferences and lessons
|
|
|
66
66
|
maggie feedback ... # redact, preview, submit, list
|
|
67
67
|
maggie localization ... # plan, validate, review, publish, stale
|
|
68
68
|
maggie service ... # import, sync, generate, validate
|
|
69
|
+
maggie seo performance ... # sampled PageSpeed/CWV report and baseline
|
|
70
|
+
maggie seo images ... # inventory, variants, confirmation, validate
|
|
71
|
+
maggie seo sitemap ... # typed plan, validate, apply, rollback
|
|
69
72
|
maggie deployment | migration | release | analytics | schedule
|
|
70
73
|
maggie api lifecycle | site-audit | ops audit
|
|
71
74
|
```
|
|
@@ -77,11 +80,38 @@ history only and is not an installable package skill.
|
|
|
77
80
|
with `--confirm`, removes only known retired EmDash artifacts. It preserves
|
|
78
81
|
`docs/de-emdash-*` migration history and `.maggie` marketplace state.
|
|
79
82
|
|
|
83
|
+
### SEO performance, images, and sitemaps
|
|
84
|
+
|
|
85
|
+
The SEO workflow keeps performance evidence separate from deterministic HTML
|
|
86
|
+
correctness checks. Performance reports use multiple samples, exclude duplicate
|
|
87
|
+
PageSpeed `fetchTime` observations, and report median/range before comparing a
|
|
88
|
+
baseline:
|
|
89
|
+
|
|
90
|
+
```bash
|
|
91
|
+
maggie seo performance --url https://example.com --strategy mobile,desktop \
|
|
92
|
+
--samples 3 --spacing-seconds 30 --output docs/performance-report.json
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
Responsive image planning emits only confirmed candidates in `srcset`, while
|
|
96
|
+
sitemap planning groups content types, emits valid empty chunk 1 files, and
|
|
97
|
+
rejects relative or off-origin `<loc>` values:
|
|
98
|
+
|
|
99
|
+
```bash
|
|
100
|
+
maggie seo images inventory --project . --output docs/image-inventory.json
|
|
101
|
+
maggie seo images plan --inventory docs/image-inventory.json --output docs/image-variants.json
|
|
102
|
+
maggie seo sitemap plan --origin https://example.com --routes-file docs/routes.tsv \
|
|
103
|
+
--output docs/sitemap-plan.json
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
Read the [performance PRD](https://github.com/TOPY-AI-LTD/ai-cmo-skills/blob/main/docs/seo-performance-cwv-prd.md)
|
|
107
|
+
and [image/sitemap PRD](https://github.com/TOPY-AI-LTD/ai-cmo-skills/blob/main/docs/image-sitemap-structure-prd.md)
|
|
108
|
+
for adapter, privacy, apply, rollback, and release gates.
|
|
109
|
+
|
|
80
110
|
Recommended upgrade sequence for the current release:
|
|
81
111
|
|
|
82
112
|
```bash
|
|
83
|
-
npx @topy-ai/maggie@0.
|
|
84
|
-
npx @topy-ai/maggie@0.
|
|
113
|
+
npx @topy-ai/maggie@0.6.0 update --project . --force
|
|
114
|
+
npx @topy-ai/maggie@0.6.0 cleanup --project .
|
|
85
115
|
```
|
|
86
116
|
|
|
87
117
|
## MaggieDash lifecycle
|
package/bin/maggie.js
CHANGED
|
@@ -109,6 +109,7 @@ Usage:
|
|
|
109
109
|
maggie api lifecycle --project PATH [--execute --allow-quota]
|
|
110
110
|
maggie memory <init|list|search|context|add|record-error|transition|export> --project PATH
|
|
111
111
|
maggie localization <plan|preview|validate|review|publish|stale|glossary> [options]
|
|
112
|
+
maggie seo performance|images|sitemap [options]
|
|
112
113
|
maggie feedback <collect|preview|submit|list> [options]
|
|
113
114
|
maggie site-audit URL [--crawl] [--languages en-GB,es-MX,ja-JP] [--check-hreflang]
|
|
114
115
|
maggie ops audit --project PATH
|
|
@@ -289,6 +290,13 @@ function service(args) {
|
|
|
289
290
|
process.exitCode = result.status ?? 1;
|
|
290
291
|
}
|
|
291
292
|
|
|
293
|
+
function seo(args) {
|
|
294
|
+
const command = args[0];
|
|
295
|
+
const scripts = { performance: "maggie_performance.py", images: "maggie_images.py", sitemap: "maggie_sitemap.py" };
|
|
296
|
+
if (!scripts[command]) throw new Error("seo command must be performance, images, or sitemap");
|
|
297
|
+
workflowCli(scripts[command], args.slice(1));
|
|
298
|
+
}
|
|
299
|
+
|
|
292
300
|
function workflowCli(name, args) {
|
|
293
301
|
const root = projectRoot(args);
|
|
294
302
|
const script = join(root, "tools", "clis", name);
|
|
@@ -365,6 +373,7 @@ try {
|
|
|
365
373
|
else if (command === "auth") workflowCli("maggie_auth.py", args);
|
|
366
374
|
else if (command === "blog") workflowCli("maggie_blog.py", args);
|
|
367
375
|
else if (command === "service") service(args);
|
|
376
|
+
else if (command === "seo") seo(args);
|
|
368
377
|
else if (command === "ops") workflowCli("maggie_ops.py", args);
|
|
369
378
|
else if (command === "deployment") workflowCli("maggie_deployment.py", args);
|
|
370
379
|
else if (command === "migration") workflowCli("maggie_migration.py", args);
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
{
|
|
2
|
+
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
3
|
+
"$id": "https://schemas.noblox.app/maggie-seo/image-variants-v1.schema.json",
|
|
4
|
+
"title": "Maggie image variants manifest",
|
|
5
|
+
"type": "object",
|
|
6
|
+
"required": ["schemaVersion", "manifestId", "createdAt", "assets", "provenance"],
|
|
7
|
+
"properties": {
|
|
8
|
+
"schemaVersion": {"const": "maggie-seo-image-variants.v1"},
|
|
9
|
+
"manifestId": {"type": "string", "pattern": "^images-[A-Za-z0-9_-]+$"},
|
|
10
|
+
"createdAt": {"type": "string", "format": "date-time"},
|
|
11
|
+
"assets": {"type": "array", "items": {"type": "object", "required": ["assetId", "source", "variants", "markupPolicy"]}},
|
|
12
|
+
"provenance": {"type": "object", "required": ["redactions"]}
|
|
13
|
+
}
|
|
14
|
+
}
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
{
|
|
2
|
+
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
3
|
+
"$id": "https://schemas.noblox.app/maggie-seo/performance-report-v1.schema.json",
|
|
4
|
+
"title": "Maggie SEO performance report",
|
|
5
|
+
"type": "object",
|
|
6
|
+
"required": ["schemaVersion", "reportId", "createdAt", "provider", "policy", "pages", "summary", "provenance"],
|
|
7
|
+
"properties": {
|
|
8
|
+
"schemaVersion": {"const": "maggie-seo-performance.v1"},
|
|
9
|
+
"reportId": {"type": "string", "pattern": "^perf-[A-Za-z0-9_-]+$"},
|
|
10
|
+
"createdAt": {"type": "string", "format": "date-time"},
|
|
11
|
+
"provider": {"const": "pagespeed-insights"},
|
|
12
|
+
"providerApiVersion": {"type": "string"},
|
|
13
|
+
"project": {"type": "object", "additionalProperties": true},
|
|
14
|
+
"policy": {"$ref": "#/$defs/policy"},
|
|
15
|
+
"pages": {"type": "array", "items": {"$ref": "#/$defs/page"}},
|
|
16
|
+
"summary": {"type": "object", "required": ["status"], "properties": {"status": {"enum": ["pass", "fail", "inconclusive"]}}, "additionalProperties": true},
|
|
17
|
+
"provenance": {"type": "object", "required": ["commands", "redactions"], "additionalProperties": true}
|
|
18
|
+
},
|
|
19
|
+
"$defs": {
|
|
20
|
+
"policy": {
|
|
21
|
+
"type": "object",
|
|
22
|
+
"required": ["strategies", "requestedSamples", "minimumIndependentSamples", "spacingSeconds", "duplicateFetchTimePolicy"],
|
|
23
|
+
"properties": {
|
|
24
|
+
"strategies": {"type": "array", "items": {"enum": ["desktop", "mobile"]}},
|
|
25
|
+
"requestedSamples": {"type": "integer", "minimum": 1},
|
|
26
|
+
"minimumIndependentSamples": {"type": "integer", "minimum": 1},
|
|
27
|
+
"spacingSeconds": {"type": "number", "minimum": 0},
|
|
28
|
+
"duplicateFetchTimePolicy": {"const": "exclude"}
|
|
29
|
+
},
|
|
30
|
+
"additionalProperties": true
|
|
31
|
+
},
|
|
32
|
+
"page": {
|
|
33
|
+
"type": "object",
|
|
34
|
+
"required": ["requestedUrl", "strategy", "samples", "aggregate", "fieldData", "comparison"],
|
|
35
|
+
"properties": {
|
|
36
|
+
"requestedUrl": {"type": "string", "format": "uri"},
|
|
37
|
+
"strategy": {"enum": ["desktop", "mobile"]},
|
|
38
|
+
"samples": {"type": "array"},
|
|
39
|
+
"aggregate": {"type": "object"},
|
|
40
|
+
"fieldData": {"type": "object"},
|
|
41
|
+
"comparison": {"type": "object"}
|
|
42
|
+
},
|
|
43
|
+
"additionalProperties": true
|
|
44
|
+
}
|
|
45
|
+
}
|
|
46
|
+
}
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
{
|
|
2
|
+
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
3
|
+
"$id": "https://schemas.noblox.app/maggie-seo/sitemap-plan-v1.schema.json",
|
|
4
|
+
"title": "Maggie structured sitemap plan",
|
|
5
|
+
"type": "object",
|
|
6
|
+
"required": ["schemaVersion", "planId", "origin", "groups", "index", "redirects", "validation"],
|
|
7
|
+
"properties": {
|
|
8
|
+
"schemaVersion": {"const": "maggie-seo-sitemap-plan.v1"},
|
|
9
|
+
"planId": {"type": "string", "pattern": "^sitemap-[A-Za-z0-9_-]+$"},
|
|
10
|
+
"origin": {"type": "string", "format": "uri"},
|
|
11
|
+
"groups": {"type": "array"},
|
|
12
|
+
"index": {"type": "object"},
|
|
13
|
+
"redirects": {"type": "array"},
|
|
14
|
+
"validation": {"type": "object", "required": ["status", "errors", "warnings"]}
|
|
15
|
+
}
|
|
16
|
+
}
|
|
@@ -19,6 +19,39 @@ voice, CTA alignment, and observed search/AI outcomes.
|
|
|
19
19
|
|
|
20
20
|
## Choose the mode
|
|
21
21
|
|
|
22
|
+
For sampled PageSpeed performance and Core Web Vitals evidence, run:
|
|
23
|
+
|
|
24
|
+
```bash
|
|
25
|
+
maggie seo performance --url https://example.com --strategy mobile,desktop \
|
|
26
|
+
--samples 3 --spacing-seconds 30 --output docs/performance-report.json
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
The performance workflow excludes duplicate provider `fetchTime` observations,
|
|
30
|
+
reports median/range, and marks a baseline comparison `inconclusive` when it
|
|
31
|
+
does not have enough independent samples. `PAGESPEED_API_KEY` is optional for
|
|
32
|
+
the provider but is never written to reports. Read the [performance PRD](https://github.com/TOPY-AI-LTD/ai-cmo-skills/blob/main/docs/seo-performance-cwv-prd.md)
|
|
33
|
+
when selecting thresholds or adding a provider.
|
|
34
|
+
|
|
35
|
+
For responsive image planning and typed sitemap planning, run:
|
|
36
|
+
|
|
37
|
+
```bash
|
|
38
|
+
maggie seo images inventory --project . --output docs/image-inventory.json
|
|
39
|
+
maggie seo images plan --inventory docs/image-inventory.json --output docs/image-variants.json
|
|
40
|
+
maggie seo images validate --manifest docs/image-variants.json
|
|
41
|
+
maggie seo sitemap plan --origin https://example.com --routes-file docs/routes.tsv \
|
|
42
|
+
--output docs/sitemap-plan.json
|
|
43
|
+
maggie seo sitemap validate --plan docs/sitemap-plan.json
|
|
44
|
+
# After an approved host adapter apply, rollback uses its exact backup manifest.
|
|
45
|
+
maggie seo sitemap rollback --backup-manifest .maggie-sitemap-backups/<plan>/backup-manifest.json \
|
|
46
|
+
--public-dir public --confirm
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
Only confirmed image variants may enter `srcset`; `apply` requires an explicit
|
|
50
|
+
confirmation and host adapter. Sitemap plans keep content types separate,
|
|
51
|
+
emit chunk 1 for empty enabled types, enforce absolute same-origin URLs, and
|
|
52
|
+
record redirects for removed sitemap files. Read the [image and sitemap PRD](https://github.com/TOPY-AI-LTD/ai-cmo-skills/blob/main/docs/image-sitemap-structure-prd.md)
|
|
53
|
+
for adapter and rollback rules.
|
|
54
|
+
|
|
22
55
|
For a deterministic technical smoke check, run:
|
|
23
56
|
|
|
24
57
|
```bash
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Inventory and validate safe responsive image variant manifests."""
|
|
3
|
+
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
import argparse
|
|
7
|
+
import json
|
|
8
|
+
import sys
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from urllib.request import Request, urlopen
|
|
11
|
+
|
|
12
|
+
sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "runtime"))
|
|
13
|
+
from maggie_media import confirm_variants, inventory_project, plan_variants, validate_manifest # noqa: E402
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def write_json(path: str, value: dict) -> None:
|
|
17
|
+
target = Path(path)
|
|
18
|
+
target.parent.mkdir(parents=True, exist_ok=True)
|
|
19
|
+
target.write_text(json.dumps(value, indent=2) + "\n", encoding="utf-8")
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def check_url(url: str) -> int:
|
|
23
|
+
request = Request(url, method="HEAD", headers={"User-Agent": "Maggie-SEO-Images/1.0"})
|
|
24
|
+
with urlopen(request, timeout=20) as response:
|
|
25
|
+
return response.status
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def main() -> int:
|
|
29
|
+
parser = argparse.ArgumentParser()
|
|
30
|
+
sub = parser.add_subparsers(dest="command", required=True)
|
|
31
|
+
inventory = sub.add_parser("inventory")
|
|
32
|
+
inventory.add_argument("--project", default=".")
|
|
33
|
+
inventory.add_argument("--output", required=True)
|
|
34
|
+
plan = sub.add_parser("plan")
|
|
35
|
+
plan.add_argument("--inventory", required=True)
|
|
36
|
+
plan.add_argument("--widths", default="320,480,768,1080,1440,1920")
|
|
37
|
+
plan.add_argument("--provider", choices=("unsplash", "cloudinary"))
|
|
38
|
+
plan.add_argument("--confirm-external", action="store_true")
|
|
39
|
+
plan.add_argument("--output", required=True)
|
|
40
|
+
confirm = sub.add_parser("confirm")
|
|
41
|
+
confirm.add_argument("--manifest", required=True)
|
|
42
|
+
confirm.add_argument("--output", required=True)
|
|
43
|
+
validate = sub.add_parser("validate")
|
|
44
|
+
validate.add_argument("--manifest", required=True)
|
|
45
|
+
apply = sub.add_parser("apply")
|
|
46
|
+
apply.add_argument("--manifest", required=True)
|
|
47
|
+
apply.add_argument("--adapter", choices=("manifest",), required=True)
|
|
48
|
+
apply.add_argument("--output", required=True)
|
|
49
|
+
apply.add_argument("--confirm", action="store_true")
|
|
50
|
+
args = parser.parse_args()
|
|
51
|
+
try:
|
|
52
|
+
if args.command == "inventory":
|
|
53
|
+
result = inventory_project(args.project)
|
|
54
|
+
write_json(args.output, result)
|
|
55
|
+
print(json.dumps({"assets": len(result["assets"]), "output": str(Path(args.output).resolve())}, indent=2))
|
|
56
|
+
return 0
|
|
57
|
+
if args.command == "plan":
|
|
58
|
+
result = plan_variants(json.loads(Path(args.inventory).read_text(encoding="utf-8")), [int(item) for item in args.widths.split(",")], args.provider)
|
|
59
|
+
write_json(args.output, result)
|
|
60
|
+
print(json.dumps({"assets": len(result["assets"]), "output": str(Path(args.output).resolve())}, indent=2))
|
|
61
|
+
return 0
|
|
62
|
+
manifest = json.loads(Path(args.manifest).read_text(encoding="utf-8"))
|
|
63
|
+
if args.command == "confirm":
|
|
64
|
+
result = confirm_variants(manifest, check_url)
|
|
65
|
+
write_json(args.output, result)
|
|
66
|
+
print(json.dumps({"errors": validate_manifest(result), "output": str(Path(args.output).resolve())}, indent=2))
|
|
67
|
+
return 0 if not validate_manifest(result) else 1
|
|
68
|
+
if args.command == "validate":
|
|
69
|
+
errors = validate_manifest(manifest)
|
|
70
|
+
print(json.dumps({"status": "pass" if not errors else "fail", "errors": errors}, indent=2))
|
|
71
|
+
return 0 if not errors else 1
|
|
72
|
+
if not args.confirm:
|
|
73
|
+
raise ValueError("apply requires --confirm")
|
|
74
|
+
errors = validate_manifest(manifest)
|
|
75
|
+
if errors:
|
|
76
|
+
raise ValueError("manifest is invalid: " + "; ".join(errors))
|
|
77
|
+
applied = {**manifest, "apply": {"adapter": args.adapter, "status": "validated-local-manifest", "publicWrite": False}}
|
|
78
|
+
write_json(args.output, applied)
|
|
79
|
+
print(json.dumps({"status": applied["apply"]["status"], "publicWrite": False, "output": str(Path(args.output).resolve())}, indent=2))
|
|
80
|
+
return 0
|
|
81
|
+
except (OSError, ValueError, TypeError, json.JSONDecodeError) as exc:
|
|
82
|
+
parser.error(str(exc))
|
|
83
|
+
return 2
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
if __name__ == "__main__":
|
|
87
|
+
raise SystemExit(main())
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Measure public pages with sampled PageSpeed evidence."""
|
|
3
|
+
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
import argparse
|
|
7
|
+
import json
|
|
8
|
+
import os
|
|
9
|
+
import sys
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
|
|
12
|
+
sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "runtime"))
|
|
13
|
+
from maggie_pagespeed import STRATEGIES, fetch_pagespeed, load_json, make_report, validate_public_url # noqa: E402
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def urls_from_args(args: argparse.Namespace) -> list[str]:
|
|
17
|
+
urls = list(args.url or [])
|
|
18
|
+
if args.url_file:
|
|
19
|
+
urls.extend(line.strip() for line in Path(args.url_file).read_text(encoding="utf-8").splitlines() if line.strip() and not line.lstrip().startswith("#"))
|
|
20
|
+
unique = list(dict.fromkeys(urls))
|
|
21
|
+
if not unique:
|
|
22
|
+
raise ValueError("at least one --url or --url-file is required")
|
|
23
|
+
for url in unique:
|
|
24
|
+
validate_public_url(url)
|
|
25
|
+
return unique
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def main() -> int:
|
|
29
|
+
parser = argparse.ArgumentParser()
|
|
30
|
+
parser.add_argument("--url", action="append")
|
|
31
|
+
parser.add_argument("--url-file")
|
|
32
|
+
parser.add_argument("--strategy", default="mobile")
|
|
33
|
+
parser.add_argument("--samples", type=int, default=3)
|
|
34
|
+
parser.add_argument("--spacing-seconds", type=float, default=30)
|
|
35
|
+
parser.add_argument("--categories", default="performance")
|
|
36
|
+
parser.add_argument("--locale")
|
|
37
|
+
parser.add_argument("--fixture", help="JSON fixture containing an array of provider responses")
|
|
38
|
+
parser.add_argument("--baseline")
|
|
39
|
+
parser.add_argument("--output")
|
|
40
|
+
parser.add_argument("--fail-on", choices=("regression", "complete"), default=None)
|
|
41
|
+
parser.add_argument("--allow-low-sample", action="store_true")
|
|
42
|
+
parser.add_argument("--dry-run", action="store_true")
|
|
43
|
+
args = parser.parse_args()
|
|
44
|
+
try:
|
|
45
|
+
urls = urls_from_args(args)
|
|
46
|
+
strategies = [item.strip() for item in args.strategy.split(",") if item.strip()]
|
|
47
|
+
if not strategies or any(item not in STRATEGIES for item in strategies):
|
|
48
|
+
raise ValueError("strategy must contain only desktop and mobile")
|
|
49
|
+
if args.samples < 1 or (args.samples < 3 and not args.allow_low_sample):
|
|
50
|
+
raise ValueError("samples must be at least 3 unless --allow-low-sample is supplied")
|
|
51
|
+
if args.spacing_seconds < 0:
|
|
52
|
+
raise ValueError("spacing-seconds must not be negative")
|
|
53
|
+
planned_calls = len(urls) * len(strategies) * args.samples
|
|
54
|
+
if args.dry_run:
|
|
55
|
+
print(json.dumps({"urls": len(urls), "strategies": strategies, "samples": args.samples, "plannedProviderCalls": planned_calls, "spacingSeconds": args.spacing_seconds}, indent=2))
|
|
56
|
+
return 0
|
|
57
|
+
fixture_values = load_json(args.fixture) if args.fixture else None
|
|
58
|
+
fixture_values = fixture_values.get("responses", fixture_values) if fixture_values is not None else None
|
|
59
|
+
fixture_index = 0
|
|
60
|
+
|
|
61
|
+
def fetcher(url: str, strategy: str) -> dict:
|
|
62
|
+
nonlocal fixture_index
|
|
63
|
+
if fixture_values is not None:
|
|
64
|
+
if not isinstance(fixture_values, list) or not fixture_values:
|
|
65
|
+
raise ValueError("fixture must contain a non-empty response array")
|
|
66
|
+
response = fixture_values[min(fixture_index, len(fixture_values) - 1)]
|
|
67
|
+
fixture_index += 1
|
|
68
|
+
return response
|
|
69
|
+
return fetch_pagespeed(url, strategy, [item.strip() for item in args.categories.split(",") if item.strip()], args.locale, os.environ.get("PAGESPEED_API_KEY"))
|
|
70
|
+
|
|
71
|
+
baseline = load_json(args.baseline) if args.baseline else None
|
|
72
|
+
report = make_report(urls, strategies, args.samples, args.spacing_seconds, fetcher, baseline)
|
|
73
|
+
if args.output:
|
|
74
|
+
output = Path(args.output)
|
|
75
|
+
output.parent.mkdir(parents=True, exist_ok=True)
|
|
76
|
+
output.write_text(json.dumps(report, indent=2) + "\n", encoding="utf-8")
|
|
77
|
+
print(json.dumps(report, indent=2))
|
|
78
|
+
if args.fail_on == "regression" and report["summary"]["status"] == "fail":
|
|
79
|
+
return 2
|
|
80
|
+
if args.fail_on == "complete" and report["summary"]["status"] != "pass":
|
|
81
|
+
return 2
|
|
82
|
+
return 0
|
|
83
|
+
except (OSError, ValueError) as exc:
|
|
84
|
+
parser.error(str(exc))
|
|
85
|
+
return 2
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
if __name__ == "__main__":
|
|
89
|
+
raise SystemExit(main())
|
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Plan and validate typed, conservative XML sitemaps."""
|
|
3
|
+
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
import argparse
|
|
7
|
+
import hashlib
|
|
8
|
+
import json
|
|
9
|
+
import sys
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
|
|
12
|
+
sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "runtime"))
|
|
13
|
+
from maggie_sitemap import build_plan, parse_routes, validate_plan_data # noqa: E402
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def write_json(path: str, value: dict) -> None:
|
|
17
|
+
target = Path(path)
|
|
18
|
+
target.parent.mkdir(parents=True, exist_ok=True)
|
|
19
|
+
target.write_text(json.dumps(value, indent=2) + "\n", encoding="utf-8")
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def main() -> int:
|
|
23
|
+
parser = argparse.ArgumentParser()
|
|
24
|
+
sub = parser.add_subparsers(dest="command", required=True)
|
|
25
|
+
plan = sub.add_parser("plan")
|
|
26
|
+
plan.add_argument("--origin", required=True)
|
|
27
|
+
plan.add_argument("--routes-file", required=True)
|
|
28
|
+
plan.add_argument("--content-types", default="pages,posts,topics")
|
|
29
|
+
plan.add_argument("--chunk-target", type=int, default=500)
|
|
30
|
+
plan.add_argument("--previous-manifest")
|
|
31
|
+
plan.add_argument("--output", required=True)
|
|
32
|
+
validate = sub.add_parser("validate")
|
|
33
|
+
validate.add_argument("--plan", required=True)
|
|
34
|
+
apply = sub.add_parser("apply")
|
|
35
|
+
apply.add_argument("--plan", required=True)
|
|
36
|
+
apply.add_argument("--public-dir", required=True)
|
|
37
|
+
apply.add_argument("--backup-dir")
|
|
38
|
+
apply.add_argument("--confirm", action="store_true")
|
|
39
|
+
rollback = sub.add_parser("rollback")
|
|
40
|
+
rollback.add_argument("--backup-manifest", required=True)
|
|
41
|
+
rollback.add_argument("--public-dir", required=True)
|
|
42
|
+
rollback.add_argument("--confirm", action="store_true")
|
|
43
|
+
args = parser.parse_args()
|
|
44
|
+
try:
|
|
45
|
+
if args.command == "plan":
|
|
46
|
+
previous = json.loads(Path(args.previous_manifest).read_text(encoding="utf-8")) if args.previous_manifest else None
|
|
47
|
+
result = build_plan(parse_routes(Path(args.routes_file).read_text(encoding="utf-8")), args.origin, {item.strip() for item in args.content_types.split(",") if item.strip()}, args.chunk_target, previous)
|
|
48
|
+
write_json(args.output, result)
|
|
49
|
+
print(json.dumps({"status": result["validation"]["status"], "groups": len(result["groups"]), "redirects": len(result["redirects"]), "output": str(Path(args.output).resolve())}, indent=2))
|
|
50
|
+
return 0 if result["validation"]["status"] == "pass" else 1
|
|
51
|
+
if args.command == "rollback":
|
|
52
|
+
if not args.confirm:
|
|
53
|
+
raise ValueError("rollback requires --confirm")
|
|
54
|
+
backup = json.loads(Path(args.backup_manifest).read_text(encoding="utf-8"))
|
|
55
|
+
public = Path(args.public_dir)
|
|
56
|
+
restored = 0
|
|
57
|
+
for item in backup.get("files", []):
|
|
58
|
+
destination = public / item["filename"]
|
|
59
|
+
source = Path(args.backup_manifest).parent / item["filename"]
|
|
60
|
+
if item["existed"]:
|
|
61
|
+
destination.write_bytes(source.read_bytes())
|
|
62
|
+
restored += 1
|
|
63
|
+
elif destination.exists() and hashlib.sha256(destination.read_bytes()).hexdigest() == item["appliedSha256"]:
|
|
64
|
+
destination.unlink()
|
|
65
|
+
restored += 1
|
|
66
|
+
print(json.dumps({"status": "rolled-back", "restored": restored}, indent=2))
|
|
67
|
+
return 0
|
|
68
|
+
plan_value = json.loads(Path(args.plan).read_text(encoding="utf-8"))
|
|
69
|
+
if args.command == "validate":
|
|
70
|
+
result = validate_plan_data(plan_value)
|
|
71
|
+
print(json.dumps(result, indent=2))
|
|
72
|
+
return 0 if result["status"] == "pass" else 1
|
|
73
|
+
if not args.confirm:
|
|
74
|
+
raise ValueError("apply requires --confirm")
|
|
75
|
+
validation = validate_plan_data(plan_value)
|
|
76
|
+
if validation["status"] != "pass":
|
|
77
|
+
raise ValueError("plan is invalid: " + "; ".join(validation["errors"]))
|
|
78
|
+
target = Path(args.public_dir)
|
|
79
|
+
target.mkdir(parents=True, exist_ok=True)
|
|
80
|
+
backup_root = Path(args.backup_dir) if args.backup_dir else target / ".maggie-sitemap-backups" / plan_value["planId"]
|
|
81
|
+
backup_root.mkdir(parents=True, exist_ok=True)
|
|
82
|
+
backup_files = []
|
|
83
|
+
output_files = [*(chunk for group in plan_value.get("groups", []) for chunk in group.get("chunks", [])), {"filename": "sitemap.xml", "xml": plan_value["index"]["xml"]}]
|
|
84
|
+
for item in output_files:
|
|
85
|
+
destination = target / item["filename"]
|
|
86
|
+
existed = destination.exists()
|
|
87
|
+
if existed:
|
|
88
|
+
(backup_root / item["filename"]).parent.mkdir(parents=True, exist_ok=True)
|
|
89
|
+
(backup_root / item["filename"]).write_bytes(destination.read_bytes())
|
|
90
|
+
content = item["xml"].encode("utf-8")
|
|
91
|
+
backup_files.append({"filename": item["filename"], "existed": existed, "appliedSha256": hashlib.sha256(content).hexdigest()})
|
|
92
|
+
for group in plan_value.get("groups", []):
|
|
93
|
+
for chunk in group.get("chunks", []):
|
|
94
|
+
(target / chunk["filename"]).write_text(chunk["xml"], encoding="utf-8")
|
|
95
|
+
(target / "sitemap.xml").write_text(plan_value["index"]["xml"], encoding="utf-8")
|
|
96
|
+
if plan_value.get("redirects"):
|
|
97
|
+
write_json(str(target / "sitemap-redirect-plan.json"), {"redirects": plan_value["redirects"]})
|
|
98
|
+
backup_manifest = {"schemaVersion": "maggie-seo-sitemap-rollback.v1", "planId": plan_value["planId"], "publicDir": str(target.resolve()), "files": backup_files}
|
|
99
|
+
write_json(str(backup_root / "backup-manifest.json"), backup_manifest)
|
|
100
|
+
print(json.dumps({"status": "applied-local-public-dir", "publicWrite": True, "files": len(output_files), "backupManifest": str((backup_root / "backup-manifest.json").resolve())}, indent=2))
|
|
101
|
+
return 0
|
|
102
|
+
except (OSError, ValueError, TypeError, json.JSONDecodeError) as exc:
|
|
103
|
+
parser.error(str(exc))
|
|
104
|
+
return 2
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
if __name__ == "__main__":
|
|
108
|
+
raise SystemExit(main())
|
|
@@ -0,0 +1,144 @@
|
|
|
1
|
+
"""Safe image inventory, responsive variant planning, and confirmation helpers."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import hashlib
|
|
6
|
+
import mimetypes
|
|
7
|
+
import re
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from typing import Any, Callable
|
|
10
|
+
from urllib.parse import parse_qsl, urlencode, urlparse, urlunparse
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
IMAGE_TYPES = {".avif", ".gif", ".jpeg", ".jpg", ".png", ".webp"}
|
|
14
|
+
SKIP_DIRS = {".git", ".maggie", "node_modules", "dist", "build", ".astro"}
|
|
15
|
+
THIRD_PARTY_PROVIDERS = {"unsplash", "cloudinary"}
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def asset_id(value: str) -> str:
|
|
19
|
+
return "asset-" + hashlib.sha256(value.encode("utf-8")).hexdigest()[:16]
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def image_dimensions(path: Path) -> tuple[int | None, int | None]:
|
|
23
|
+
try:
|
|
24
|
+
data = path.read_bytes()
|
|
25
|
+
except OSError:
|
|
26
|
+
return None, None
|
|
27
|
+
if data.startswith(b"\x89PNG\r\n\x1a\n") and len(data) >= 24:
|
|
28
|
+
return int.from_bytes(data[16:20], "big"), int.from_bytes(data[20:24], "big")
|
|
29
|
+
if data.startswith(b"RIFF") and data[8:12] == b"WEBP" and len(data) >= 30:
|
|
30
|
+
if data[12:16] == b"VP8X":
|
|
31
|
+
return 1 + int.from_bytes(data[24:27], "little"), 1 + int.from_bytes(data[27:30], "little")
|
|
32
|
+
if data.startswith(b"\xff\xd8"):
|
|
33
|
+
index = 2
|
|
34
|
+
while index + 9 < len(data):
|
|
35
|
+
if data[index] != 0xFF:
|
|
36
|
+
index += 1
|
|
37
|
+
continue
|
|
38
|
+
marker = data[index + 1]
|
|
39
|
+
index += 2
|
|
40
|
+
if marker in {0xD8, 0xD9}:
|
|
41
|
+
continue
|
|
42
|
+
length = int.from_bytes(data[index:index + 2], "big")
|
|
43
|
+
if marker in set(range(0xC0, 0xC4)) | set(range(0xC5, 0xC8)) | set(range(0xC9, 0xCC)) | set(range(0xCD, 0xD0)):
|
|
44
|
+
return int.from_bytes(data[index + 5:index + 7], "big"), int.from_bytes(data[index + 3:index + 5], "big")
|
|
45
|
+
index += length
|
|
46
|
+
return None, None
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def inventory_project(project: str | Path) -> dict[str, Any]:
|
|
50
|
+
root = Path(project).resolve()
|
|
51
|
+
assets = []
|
|
52
|
+
for path in sorted(root.rglob("*")):
|
|
53
|
+
if not path.is_file() or path.suffix.lower() not in IMAGE_TYPES or any(part in SKIP_DIRS for part in path.parts):
|
|
54
|
+
continue
|
|
55
|
+
relative = path.relative_to(root).as_posix()
|
|
56
|
+
width, height = image_dimensions(path)
|
|
57
|
+
assets.append({"assetId": asset_id(relative), "source": {"path": relative, "visibility": "public" if relative.startswith("public/") else "unknown"}, "mimeType": mimetypes.guess_type(path.name)[0] or "application/octet-stream", "bytes": path.stat().st_size, "intrinsic": {"width": width, "height": height}, "altText": {"status": "unknown"}})
|
|
58
|
+
return {"schemaVersion": "maggie-seo-image-inventory.v1", "assets": assets, "provenance": {"project": str(root), "redactions": ["absolute_project_path_in_public_reports"]}}
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def transform_third_party_url(url: str, provider: str, width: int) -> str:
|
|
62
|
+
if provider not in THIRD_PARTY_PROVIDERS:
|
|
63
|
+
raise ValueError("third-party provider is not allowlisted")
|
|
64
|
+
parsed = urlparse(url)
|
|
65
|
+
if parsed.scheme not in {"http", "https"} or not parsed.hostname:
|
|
66
|
+
raise ValueError("third-party source must be an absolute HTTP(S) URL")
|
|
67
|
+
if parsed.username or parsed.password or parsed.fragment:
|
|
68
|
+
raise ValueError("third-party source contains unsafe URL data")
|
|
69
|
+
if width < 1:
|
|
70
|
+
raise ValueError("variant width must be positive")
|
|
71
|
+
if provider == "unsplash":
|
|
72
|
+
query = dict(parse_qsl(parsed.query, keep_blank_values=True))
|
|
73
|
+
query.update({"w": str(width), "auto": "format"})
|
|
74
|
+
return urlunparse((parsed.scheme, parsed.netloc, parsed.path, parsed.params, urlencode(query), ""))
|
|
75
|
+
if "/upload/" not in parsed.path:
|
|
76
|
+
raise ValueError("cloudinary URL must contain an /upload/ segment")
|
|
77
|
+
path = parsed.path.replace("/upload/", f"/upload/w_{width},f_auto/", 1)
|
|
78
|
+
return urlunparse((parsed.scheme, parsed.netloc, path, parsed.params, parsed.query, ""))
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def plan_variants(inventory: dict[str, Any], widths: list[int], provider: str | None = None) -> dict[str, Any]:
|
|
82
|
+
if not widths or any(width < 1 for width in widths):
|
|
83
|
+
raise ValueError("widths must contain positive integers")
|
|
84
|
+
assets = []
|
|
85
|
+
for item in inventory.get("assets", []):
|
|
86
|
+
source = item.get("source", {})
|
|
87
|
+
source_url = source.get("url")
|
|
88
|
+
source_path = source.get("path")
|
|
89
|
+
variants = []
|
|
90
|
+
for width in sorted(set(widths)):
|
|
91
|
+
if source_url and provider:
|
|
92
|
+
url = transform_third_party_url(source_url, provider, width)
|
|
93
|
+
status = "planned"
|
|
94
|
+
elif source_path:
|
|
95
|
+
path = Path(source_path)
|
|
96
|
+
url = str(path.with_name(f"{path.stem}-{width}{path.suffix}"))
|
|
97
|
+
# Planning never confirms a derived file merely because the
|
|
98
|
+
# original has the same intrinsic width. Confirmation requires
|
|
99
|
+
# checking the candidate artifact itself.
|
|
100
|
+
status = "planned"
|
|
101
|
+
else:
|
|
102
|
+
continue
|
|
103
|
+
variants.append({"width": width, "format": Path(urlparse(url).path).suffix.lstrip(".") or "original", "url": url, "status": status, "evidence": {}})
|
|
104
|
+
assets.append({**item, "variants": variants, "markupPolicy": {"allowedWidths": [item["width"] for item in variants if item["status"] == "confirmed"], "missingVariantPolicy": "omit", "fetchPriority": "auto"}})
|
|
105
|
+
return {"schemaVersion": "maggie-seo-image-variants.v1", "manifestId": "images-" + hashlib.sha256(str(len(assets)).encode()).hexdigest()[:12], "createdAt": __import__("datetime").datetime.now(__import__("datetime").timezone.utc).isoformat().replace("+00:00", "Z"), "assets": assets, "provenance": {"redactions": ["private_paths", "credentials"]}}
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def confirm_variants(manifest: dict[str, Any], checker: Callable[[str], int]) -> dict[str, Any]:
|
|
109
|
+
result = {**manifest, "assets": []}
|
|
110
|
+
for asset in manifest.get("assets", []):
|
|
111
|
+
updated = {**asset, "variants": []}
|
|
112
|
+
for variant in asset.get("variants", []):
|
|
113
|
+
candidate = {**variant}
|
|
114
|
+
if candidate.get("status") == "planned":
|
|
115
|
+
try:
|
|
116
|
+
status = checker(candidate["url"])
|
|
117
|
+
candidate["status"] = "confirmed" if status == 200 else "failed"
|
|
118
|
+
candidate["evidence"] = {"status": status}
|
|
119
|
+
except Exception as exc:
|
|
120
|
+
candidate["status"] = "failed"
|
|
121
|
+
candidate["evidence"] = {"error": type(exc).__name__}
|
|
122
|
+
updated["variants"].append(candidate)
|
|
123
|
+
updated["markupPolicy"] = {**asset.get("markupPolicy", {}), "allowedWidths": [v["width"] for v in updated["variants"] if v.get("status") == "confirmed"]}
|
|
124
|
+
result["assets"].append(updated)
|
|
125
|
+
return result
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def confirmed_srcset(asset: dict[str, Any]) -> str:
|
|
129
|
+
return ", ".join(f"{variant['url']} {variant['width']}w" for variant in asset.get("variants", []) if variant.get("status") == "confirmed")
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def validate_manifest(manifest: dict[str, Any]) -> list[str]:
|
|
133
|
+
errors = []
|
|
134
|
+
for asset in manifest.get("assets", []):
|
|
135
|
+
for variant in asset.get("variants", []):
|
|
136
|
+
if variant.get("status") == "confirmed" and not variant.get("evidence"):
|
|
137
|
+
errors.append(f"confirmed variant lacks evidence: {asset.get('assetId')}")
|
|
138
|
+
if variant.get("status") == "confirmed" and not variant.get("url"):
|
|
139
|
+
errors.append(f"confirmed variant lacks URL: {asset.get('assetId')}")
|
|
140
|
+
allowed = set(asset.get("markupPolicy", {}).get("allowedWidths", []))
|
|
141
|
+
actual = {variant.get("width") for variant in asset.get("variants", []) if variant.get("status") == "confirmed"}
|
|
142
|
+
if allowed != actual:
|
|
143
|
+
errors.append(f"markup allowedWidths does not match confirmed variants: {asset.get('assetId')}")
|
|
144
|
+
return errors
|
|
@@ -0,0 +1,213 @@
|
|
|
1
|
+
"""Provider-neutral PageSpeed sampling, normalization, and comparison helpers."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
import os
|
|
7
|
+
import re
|
|
8
|
+
import statistics
|
|
9
|
+
import time
|
|
10
|
+
from datetime import datetime, timezone
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
from typing import Any, Callable
|
|
13
|
+
from urllib.parse import parse_qsl, urlencode, urlparse, urlunparse
|
|
14
|
+
from urllib.request import Request, urlopen
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
METRICS = ("performanceScore", "lcp", "cls", "inp", "tbt", "fcp", "ttfb")
|
|
18
|
+
STRATEGIES = ("desktop", "mobile")
|
|
19
|
+
DEFAULT_THRESHOLDS = {
|
|
20
|
+
"performanceScore": {"direction": "decrease", "absolute": 0.05},
|
|
21
|
+
"lcp": {"direction": "increase", "relative": 0.10},
|
|
22
|
+
"cls": {"direction": "increase", "absolute": 0.05},
|
|
23
|
+
"inp": {"direction": "increase", "relative": 0.10},
|
|
24
|
+
"tbt": {"direction": "increase", "relative": 0.10},
|
|
25
|
+
"fcp": {"direction": "increase", "relative": 0.10},
|
|
26
|
+
"ttfb": {"direction": "increase", "relative": 0.10},
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class PageSpeedError(RuntimeError):
|
|
31
|
+
"""A safe, typed provider failure without credentials or response bodies."""
|
|
32
|
+
|
|
33
|
+
def __init__(self, kind: str, message: str) -> None:
|
|
34
|
+
self.kind = kind
|
|
35
|
+
super().__init__(message)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def now_iso() -> str:
|
|
39
|
+
return datetime.now(timezone.utc).isoformat().replace("+00:00", "Z")
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def validate_public_url(value: str) -> str:
|
|
43
|
+
parsed = urlparse(value.strip())
|
|
44
|
+
if parsed.scheme not in {"http", "https"} or not parsed.hostname:
|
|
45
|
+
raise ValueError("URL must be an absolute HTTP(S) URL")
|
|
46
|
+
if parsed.username or parsed.password:
|
|
47
|
+
raise ValueError("URL must not contain credentials")
|
|
48
|
+
if parsed.fragment:
|
|
49
|
+
raise ValueError("URL must not contain a fragment")
|
|
50
|
+
return value.strip()
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def redact_url(value: str) -> str:
|
|
54
|
+
parsed = urlparse(value)
|
|
55
|
+
safe_query = [(key, "[redacted]") if re.search(r"token|key|secret|password", key, re.I) else (key, val) for key, val in parse_qsl(parsed.query, keep_blank_values=True)]
|
|
56
|
+
return urlunparse((parsed.scheme, parsed.netloc, parsed.path, parsed.params, urlencode(safe_query), ""))
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _number(value: Any) -> float | None:
|
|
60
|
+
return float(value) if isinstance(value, (int, float)) and not isinstance(value, bool) else None
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _audit_number(audits: dict[str, Any], *names: str) -> float | None:
|
|
64
|
+
for name in names:
|
|
65
|
+
value = audits.get(name, {})
|
|
66
|
+
if isinstance(value, dict):
|
|
67
|
+
number = _number(value.get("numericValue"))
|
|
68
|
+
if number is not None:
|
|
69
|
+
return number
|
|
70
|
+
return None
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _field_metric(metrics: dict[str, Any], *names: str) -> dict[str, Any]:
|
|
74
|
+
for name in names:
|
|
75
|
+
value = metrics.get(name)
|
|
76
|
+
if isinstance(value, dict):
|
|
77
|
+
percentile = _number(value.get("percentile"))
|
|
78
|
+
return {"value": percentile, "category": value.get("category"), "source": "field"}
|
|
79
|
+
return {"value": None, "source": "field", "status": "not_available"}
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def normalize_response(response: dict[str, Any], requested_url: str, strategy: str, sampled_at: str | None = None) -> dict[str, Any]:
|
|
83
|
+
lighthouse = response.get("lighthouseResult") or {}
|
|
84
|
+
audits = lighthouse.get("audits") if isinstance(lighthouse.get("audits"), dict) else {}
|
|
85
|
+
categories = lighthouse.get("categories") if isinstance(lighthouse.get("categories"), dict) else {}
|
|
86
|
+
performance = categories.get("performance") if isinstance(categories.get("performance"), dict) else {}
|
|
87
|
+
loading = response.get("loadingExperience") if isinstance(response.get("loadingExperience"), dict) else {}
|
|
88
|
+
field_metrics = loading.get("metrics") if isinstance(loading.get("metrics"), dict) else {}
|
|
89
|
+
lab = {
|
|
90
|
+
"performanceScore": _number(performance.get("score")),
|
|
91
|
+
"lcp": _audit_number(audits, "largest-contentful-paint"),
|
|
92
|
+
"cls": _audit_number(audits, "cumulative-layout-shift"),
|
|
93
|
+
"inp": _audit_number(audits, "interaction-to-next-paint", "experimental-interaction-to-next-paint"),
|
|
94
|
+
"tbt": _audit_number(audits, "total-blocking-time"),
|
|
95
|
+
"fcp": _audit_number(audits, "first-contentful-paint"),
|
|
96
|
+
"ttfb": _audit_number(audits, "server-response-time", "time-to-first-byte"),
|
|
97
|
+
}
|
|
98
|
+
field = {
|
|
99
|
+
"lcp": _field_metric(field_metrics, "LARGEST_CONTENTFUL_PAINT_MS"),
|
|
100
|
+
"cls": _field_metric(field_metrics, "CUMULATIVE_LAYOUT_SHIFT_SCORE"),
|
|
101
|
+
"inp": _field_metric(field_metrics, "INTERACTION_TO_NEXT_PAINT"),
|
|
102
|
+
}
|
|
103
|
+
return {
|
|
104
|
+
"requestedUrl": redact_url(requested_url),
|
|
105
|
+
"finalUrl": redact_url(str(lighthouse.get("finalUrl") or response.get("id") or requested_url)),
|
|
106
|
+
"strategy": strategy,
|
|
107
|
+
"sampledAt": sampled_at or now_iso(),
|
|
108
|
+
"fetchTime": lighthouse.get("fetchTime"),
|
|
109
|
+
"duplicateFetchTime": False,
|
|
110
|
+
"status": "ok",
|
|
111
|
+
"providerVersion": lighthouse.get("lighthouseVersion"),
|
|
112
|
+
"warnings": lighthouse.get("runWarnings") if isinstance(lighthouse.get("runWarnings"), list) else [],
|
|
113
|
+
"lab": lab,
|
|
114
|
+
"fieldData": {
|
|
115
|
+
"status": "available" if field_metrics else "not_available",
|
|
116
|
+
"source": "pagespeed-loading-experience",
|
|
117
|
+
"metrics": field,
|
|
118
|
+
},
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def fetch_pagespeed(url: str, strategy: str, categories: list[str] | None = None, locale: str | None = None, api_key: str | None = None, timeout: int = 45) -> dict[str, Any]:
|
|
123
|
+
validate_public_url(url)
|
|
124
|
+
if strategy not in STRATEGIES:
|
|
125
|
+
raise ValueError("strategy must be desktop or mobile")
|
|
126
|
+
query = {"url": url, "strategy": strategy}
|
|
127
|
+
for category in categories or ["performance"]:
|
|
128
|
+
query.setdefault("category", category)
|
|
129
|
+
if locale:
|
|
130
|
+
query["locale"] = locale
|
|
131
|
+
if api_key:
|
|
132
|
+
query["key"] = api_key
|
|
133
|
+
endpoint = "https://pagespeedonline.googleapis.com/pagespeedonline/v5/runPagespeed?" + urlencode(query)
|
|
134
|
+
request = Request(endpoint, headers={"User-Agent": "Maggie-SEO-Performance/1.0", "Accept": "application/json"})
|
|
135
|
+
try:
|
|
136
|
+
with urlopen(request, timeout=timeout) as response:
|
|
137
|
+
if response.status != 200:
|
|
138
|
+
raise PageSpeedError(f"http_{response.status}", f"PageSpeed returned HTTP {response.status}")
|
|
139
|
+
return json.loads(response.read(5_000_000).decode("utf-8"))
|
|
140
|
+
except PageSpeedError:
|
|
141
|
+
raise
|
|
142
|
+
except TimeoutError as exc:
|
|
143
|
+
raise PageSpeedError("timeout", "PageSpeed request timed out") from exc
|
|
144
|
+
except json.JSONDecodeError as exc:
|
|
145
|
+
raise PageSpeedError("invalid_json", "PageSpeed returned invalid JSON") from exc
|
|
146
|
+
except Exception as exc:
|
|
147
|
+
status = getattr(exc, "code", None)
|
|
148
|
+
if status:
|
|
149
|
+
raise PageSpeedError(f"http_{status}", f"PageSpeed returned HTTP {status}") from exc
|
|
150
|
+
raise PageSpeedError("network", f"PageSpeed request failed: {type(exc).__name__}") from exc
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def aggregate_samples(samples: list[dict[str, Any]], minimum: int = 3) -> dict[str, Any]:
|
|
154
|
+
independent = [sample for sample in samples if sample.get("status") == "ok" and not sample.get("duplicateFetchTime")]
|
|
155
|
+
aggregate: dict[str, Any] = {"independentSampleCount": len(independent), "status": "pass" if len(independent) >= minimum else "inconclusive", "metrics": {}}
|
|
156
|
+
for metric in METRICS:
|
|
157
|
+
values = [_number((sample.get("lab") or {}).get(metric)) for sample in independent]
|
|
158
|
+
values = [value for value in values if value is not None]
|
|
159
|
+
aggregate["metrics"][metric] = {"median": statistics.median(values) if values else None, "min": min(values) if values else None, "max": max(values) if values else None, "spread": max(values) - min(values) if values else None, "sampleCount": len(values), "status": "available" if values else "not_available"}
|
|
160
|
+
return aggregate
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def compare_aggregate(current: dict[str, Any], baseline: dict[str, Any] | None, thresholds: dict[str, dict[str, float]] | None = None) -> dict[str, Any]:
|
|
164
|
+
if not baseline:
|
|
165
|
+
return {"status": "no_baseline", "baselineId": None, "changes": [], "reasons": ["no baseline supplied"]}
|
|
166
|
+
if current.get("independentSampleCount", 0) < 3 or baseline.get("independentSampleCount", 0) < 3:
|
|
167
|
+
return {"status": "inconclusive", "baselineId": baseline.get("reportId"), "changes": [], "reasons": ["both current and baseline require at least three independent samples"]}
|
|
168
|
+
policy = thresholds or DEFAULT_THRESHOLDS
|
|
169
|
+
changes = []
|
|
170
|
+
for metric, rule in policy.items():
|
|
171
|
+
before = _number((baseline.get("metrics", {}).get(metric) or {}).get("median"))
|
|
172
|
+
after = _number((current.get("metrics", {}).get(metric) or {}).get("median"))
|
|
173
|
+
if before is None or after is None:
|
|
174
|
+
continue
|
|
175
|
+
delta = after - before
|
|
176
|
+
breach = delta <= -rule.get("absolute", float("inf")) if rule["direction"] == "decrease" and "absolute" in rule else delta >= rule.get("absolute", float("inf")) if rule["direction"] == "increase" and "absolute" in rule else (after <= before * (1 - rule.get("relative", 0)) if rule["direction"] == "decrease" else after >= before * (1 + rule.get("relative", 0)))
|
|
177
|
+
changes.append({"metric": metric, "before": before, "after": after, "delta": delta, "breach": bool(breach), "rule": rule})
|
|
178
|
+
breached = [change for change in changes if change["breach"]]
|
|
179
|
+
return {"status": "regression" if len(breached) >= 1 else "no_regression", "baselineId": baseline.get("reportId"), "changes": changes, "reasons": [f"{item['metric']} exceeded configured threshold" for item in breached]}
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def make_report(urls: list[str], strategies: list[str], samples: int, spacing_seconds: float, fetcher: Callable[[str, str], dict[str, Any]], baseline: dict[str, Any] | None = None, minimum: int = 3) -> dict[str, Any]:
|
|
183
|
+
pages = []
|
|
184
|
+
for url in urls:
|
|
185
|
+
validate_public_url(url)
|
|
186
|
+
for strategy in strategies:
|
|
187
|
+
records = []
|
|
188
|
+
seen_fetch_times: set[str] = set()
|
|
189
|
+
for index in range(samples):
|
|
190
|
+
if index and spacing_seconds:
|
|
191
|
+
time.sleep(spacing_seconds)
|
|
192
|
+
try:
|
|
193
|
+
record = normalize_response(fetcher(url, strategy), url, strategy)
|
|
194
|
+
fetch_time = record.get("fetchTime")
|
|
195
|
+
if fetch_time and fetch_time in seen_fetch_times:
|
|
196
|
+
record["duplicateFetchTime"] = True
|
|
197
|
+
elif fetch_time:
|
|
198
|
+
seen_fetch_times.add(fetch_time)
|
|
199
|
+
except PageSpeedError as exc:
|
|
200
|
+
record = {"requestedUrl": redact_url(url), "strategy": strategy, "sampledAt": now_iso(), "status": "error", "error": {"kind": exc.kind, "message": str(exc)}}
|
|
201
|
+
records.append(record)
|
|
202
|
+
aggregate = aggregate_samples(records, minimum)
|
|
203
|
+
baseline_page = None
|
|
204
|
+
if baseline:
|
|
205
|
+
baseline_page = next((item.get("aggregate") for item in baseline.get("pages", []) if item.get("requestedUrl") == redact_url(url) and item.get("strategy") == strategy), None)
|
|
206
|
+
pages.append({"requestedUrl": redact_url(url), "strategy": strategy, "samples": records, "aggregate": aggregate, "fieldData": {"status": "available" if any((item.get("fieldData") or {}).get("status") == "available" for item in records) else "not_available"}, "comparison": compare_aggregate(aggregate, baseline_page)})
|
|
207
|
+
statuses = [page["comparison"]["status"] for page in pages]
|
|
208
|
+
summary = "fail" if "regression" in statuses else "inconclusive" if any(status == "inconclusive" or page["aggregate"]["status"] == "inconclusive" for status, page in zip(statuses, pages)) else "pass"
|
|
209
|
+
return {"schemaVersion": "maggie-seo-performance.v1", "reportId": "perf-" + datetime.now(timezone.utc).strftime("%Y%m%d%H%M%S"), "createdAt": now_iso(), "provider": "pagespeed-insights", "providerApiVersion": "v5", "policy": {"strategies": strategies, "requestedSamples": samples, "minimumIndependentSamples": minimum, "spacingSeconds": spacing_seconds, "duplicateFetchTimePolicy": "exclude"}, "pages": pages, "summary": {"status": summary}, "provenance": {"commands": [], "redactions": ["api_key", "credentialed_urls"]}}
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def load_json(path: str | Path) -> dict[str, Any]:
|
|
213
|
+
return json.loads(Path(path).read_text(encoding="utf-8"))
|
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
"""Typed sitemap planning, chunking, validation, and redirect planning."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import hashlib
|
|
6
|
+
import re
|
|
7
|
+
from datetime import datetime, timezone
|
|
8
|
+
from html import escape
|
|
9
|
+
from urllib.parse import parse_qs, urlparse
|
|
10
|
+
from xml.etree import ElementTree
|
|
11
|
+
|
|
12
|
+
HARD_URL_LIMIT = 50_000
|
|
13
|
+
HARD_BYTES_LIMIT = 52_428_800
|
|
14
|
+
DEFAULT_CHUNK_TARGET = 500
|
|
15
|
+
EXCLUDED_PATHS = re.compile(r"/(search|find|login|draft|preview)(/|$)", re.I)
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def absolute_url(url: str, origin: str) -> bool:
|
|
19
|
+
parsed = urlparse(url)
|
|
20
|
+
return parsed.scheme in {"http", "https"} and bool(parsed.netloc) and not parsed.fragment and parsed.netloc == urlparse(origin).netloc
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def parse_routes(text: str) -> list[dict[str, str]]:
|
|
24
|
+
routes = []
|
|
25
|
+
for line in text.splitlines():
|
|
26
|
+
if not line.strip() or line.lstrip().startswith("#"):
|
|
27
|
+
continue
|
|
28
|
+
parts = line.split("\t")
|
|
29
|
+
if len(parts) < 2:
|
|
30
|
+
raise ValueError("route lines must be type<TAB>absolute-url<TAB>optional-lastmod")
|
|
31
|
+
item = {"type": parts[0].strip(), "url": parts[1].strip()}
|
|
32
|
+
if len(parts) > 2 and parts[2].strip():
|
|
33
|
+
item["lastmod"] = parts[2].strip()
|
|
34
|
+
routes.append(item)
|
|
35
|
+
return routes
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def filter_routes(routes: list[dict[str, str]], origin: str, content_types: set[str]) -> tuple[dict[str, list[dict[str, str]]], list[dict[str, str]]]:
|
|
39
|
+
groups = {item: [] for item in sorted(content_types)}
|
|
40
|
+
excluded = []
|
|
41
|
+
for route in routes:
|
|
42
|
+
url = route.get("url", "")
|
|
43
|
+
parsed = urlparse(url)
|
|
44
|
+
reason = None
|
|
45
|
+
if route.get("type") not in content_types:
|
|
46
|
+
reason = "content type not enabled"
|
|
47
|
+
elif not absolute_url(url, origin):
|
|
48
|
+
reason = "URL must be an absolute HTTP(S) URL on the approved origin"
|
|
49
|
+
elif EXCLUDED_PATHS.search(parsed.path) or any(key in parse_qs(parsed.query) for key in ("q", "search", "filter", "page")):
|
|
50
|
+
reason = "search, filter, or non-canonical route"
|
|
51
|
+
elif route.get("type", "").lower() in {"rss", "atom", "feed"}:
|
|
52
|
+
reason = "feed is not an HTML sitemap entry"
|
|
53
|
+
if reason:
|
|
54
|
+
excluded.append({"url": url, "type": route.get("type", ""), "reason": reason})
|
|
55
|
+
else:
|
|
56
|
+
groups.setdefault(route["type"], []).append(route)
|
|
57
|
+
for key in groups:
|
|
58
|
+
groups[key] = list({item["url"]: item for item in groups[key]}.values())
|
|
59
|
+
return groups, excluded
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def xml_file(urls: list[dict[str, str]]) -> str:
|
|
63
|
+
rows = []
|
|
64
|
+
for item in urls:
|
|
65
|
+
lastmod = f"<lastmod>{escape(item['lastmod'])}</lastmod>" if item.get("lastmod") else ""
|
|
66
|
+
rows.append(f"<url><loc>{escape(item['url'])}</loc>{lastmod}</url>")
|
|
67
|
+
return '<?xml version="1.0" encoding="UTF-8"?>\n<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">' + "".join(rows) + "</urlset>\n"
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def build_plan(routes: list[dict[str, str]], origin: str, content_types: set[str], chunk_target: int = DEFAULT_CHUNK_TARGET, previous: dict | None = None) -> dict:
|
|
71
|
+
if chunk_target < 1 or chunk_target > HARD_URL_LIMIT:
|
|
72
|
+
raise ValueError("chunk target must be between 1 and 50000")
|
|
73
|
+
groups, excluded = filter_routes(routes, origin, content_types)
|
|
74
|
+
group_plans = []
|
|
75
|
+
sitemap_urls = []
|
|
76
|
+
for content_type in sorted(content_types):
|
|
77
|
+
chunks = []
|
|
78
|
+
entries = groups.get(content_type, []) or []
|
|
79
|
+
for index in range(0, max(len(entries), 1), chunk_target):
|
|
80
|
+
chunk_entries = entries[index:index + chunk_target]
|
|
81
|
+
filename = f"sitemap-{content_type}-{index // chunk_target + 1}.xml"
|
|
82
|
+
xml = xml_file(chunk_entries)
|
|
83
|
+
url = origin.rstrip("/") + "/" + filename
|
|
84
|
+
chunks.append({"filename": filename, "url": url, "entries": len(chunk_entries), "bytes": len(xml.encode()), "sha256": hashlib.sha256(xml.encode()).hexdigest(), "xml": xml})
|
|
85
|
+
sitemap_urls.append(url)
|
|
86
|
+
group_plans.append({"contentType": content_type, "chunks": chunks})
|
|
87
|
+
index_xml = '<?xml version="1.0" encoding="UTF-8"?>\n<sitemapindex xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">' + "".join(f"<sitemap><loc>{escape(url)}</loc></sitemap>" for url in sitemap_urls) + "</sitemapindex>\n"
|
|
88
|
+
current = set(sitemap_urls)
|
|
89
|
+
old = set((previous or {}).get("sitemapUrls", []))
|
|
90
|
+
redirects = [{"from": url, "to": origin.rstrip("/") + "/sitemap.xml", "reason": "previously advertised sitemap removed"} for url in sorted(old - current)]
|
|
91
|
+
plan = {"schemaVersion": "maggie-seo-sitemap-plan.v1", "planId": "sitemap-" + datetime.now(timezone.utc).strftime("%Y%m%d%H%M%S"), "origin": origin.rstrip("/"), "groups": group_plans, "index": {"url": origin.rstrip("/") + "/sitemap.xml", "bytes": len(index_xml.encode()), "xml": index_xml, "sitemapUrls": sitemap_urls}, "redirects": redirects, "excluded": excluded}
|
|
92
|
+
plan["validation"] = validate_plan_data(plan)
|
|
93
|
+
return plan
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def validate_plan_data(plan: dict) -> dict:
|
|
97
|
+
errors = []
|
|
98
|
+
warnings = []
|
|
99
|
+
origin = plan.get("origin", "")
|
|
100
|
+
if not urlparse(origin).scheme or not urlparse(origin).netloc:
|
|
101
|
+
errors.append("origin must be an absolute HTTP(S) URL")
|
|
102
|
+
expected_urls = []
|
|
103
|
+
for group in plan.get("groups", []):
|
|
104
|
+
for chunk in group.get("chunks", []):
|
|
105
|
+
expected_urls.append(chunk.get("url"))
|
|
106
|
+
if chunk.get("entries", 0) > HARD_URL_LIMIT:
|
|
107
|
+
errors.append(f"chunk exceeds URL limit: {chunk.get('filename')}")
|
|
108
|
+
if chunk.get("bytes", 0) > HARD_BYTES_LIMIT:
|
|
109
|
+
errors.append(f"chunk exceeds byte limit: {chunk.get('filename')}")
|
|
110
|
+
try:
|
|
111
|
+
root = ElementTree.fromstring(chunk.get("xml", ""))
|
|
112
|
+
locs = [element.text or "" for element in root.iter() if element.tag.endswith("loc")]
|
|
113
|
+
if any(not absolute_url(value, origin) for value in locs):
|
|
114
|
+
errors.append(f"chunk contains a relative or off-origin loc: {chunk.get('filename')}")
|
|
115
|
+
except ElementTree.ParseError:
|
|
116
|
+
errors.append(f"chunk is not valid XML: {chunk.get('filename')}")
|
|
117
|
+
index = plan.get("index", {})
|
|
118
|
+
if index.get("bytes", 0) > HARD_BYTES_LIMIT:
|
|
119
|
+
errors.append("sitemap index exceeds byte limit")
|
|
120
|
+
try:
|
|
121
|
+
index_root = ElementTree.fromstring(index.get("xml", ""))
|
|
122
|
+
index_locs = [element.text or "" for element in index_root.iter() if element.tag.endswith("loc")]
|
|
123
|
+
if any(not absolute_url(value, origin) for value in index_locs):
|
|
124
|
+
errors.append("sitemap index contains a relative or off-origin loc")
|
|
125
|
+
if set(index_locs) != set(expected_urls):
|
|
126
|
+
errors.append("sitemap index references do not match generated chunks")
|
|
127
|
+
except ElementTree.ParseError:
|
|
128
|
+
errors.append("sitemap index is not valid XML")
|
|
129
|
+
if not expected_urls:
|
|
130
|
+
warnings.append("no sitemap chunks generated")
|
|
131
|
+
return {"status": "pass" if not errors else "fail", "errors": errors, "warnings": warnings}
|