@topy-ai/maggie 0.6.7 → 0.6.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -2
- package/README.zh-TW.md +1 -1
- package/bundled-skills/maggie-content-localization/SKILL.md +1 -1
- package/bundled-skills/maggie-deployment/SKILL.md +23 -1
- package/bundled-skills/maggie-design/SKILL.md +1 -1
- package/bundled-skills/maggie-seo-geo/SKILL.md +1 -1
- package/bundled-skills/maggie-service-booking/SKILL.md +1 -1
- package/bundled-tools/clis/maggie_deployment.py +60 -2
- package/bundled-tools/clis/maggie_design.py +12 -0
- package/bundled-tools/clis/maggie_localization.py +11 -0
- package/bundled-tools/clis/maggie_service_booking.py +3 -3
- package/bundled-tools/clis/site_audit.py +12 -1
- package/bundled-tools/runtime/maggie_sitemap.py +16 -3
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -179,8 +179,8 @@ artifact schemas.
|
|
|
179
179
|
Recommended upgrade sequence for the current release:
|
|
180
180
|
|
|
181
181
|
```bash
|
|
182
|
-
npx @topy-ai/maggie@0.6.
|
|
183
|
-
npx @topy-ai/maggie@0.6.
|
|
182
|
+
npx @topy-ai/maggie@0.6.9 update --project . --force
|
|
183
|
+
npx @topy-ai/maggie@0.6.9 cleanup --project .
|
|
184
184
|
```
|
|
185
185
|
|
|
186
186
|
## MaggieDash lifecycle
|
package/README.zh-TW.md
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
name: maggie-content-localization
|
|
3
3
|
description: Manage translation, polish, rewrite, market localization, review, stale detection, and publishing for pages, guides, posts, services, products, and categories.
|
|
4
4
|
metadata:
|
|
5
|
-
version: 1.
|
|
5
|
+
version: 1.3.0
|
|
6
6
|
---
|
|
7
7
|
|
|
8
8
|
# Maggie Content Localization
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
name: maggie-deployment
|
|
3
3
|
description: Deploy and operate Maggie blog projects with Cloudflare Workers as the default target, while preserving an adapter boundary for VPS, GCP, and AWS.
|
|
4
4
|
metadata:
|
|
5
|
-
version: 1.
|
|
5
|
+
version: 1.3.0
|
|
6
6
|
---
|
|
7
7
|
|
|
8
8
|
# Maggie Deployment
|
|
@@ -70,6 +70,20 @@ The output contains only the immutable release/current layout, systemd unit,
|
|
|
70
70
|
Nginx reverse-proxy config, health commands and rollback command. It never
|
|
71
71
|
contains credentials and never SSHs, changes DNS, restarts systemd or deploys.
|
|
72
72
|
|
|
73
|
+
VPS plans include a bounded retention policy: keep two immutable releases,
|
|
74
|
+
preserve the `current` target and rollback target, and review prune candidates
|
|
75
|
+
before any operator executes cleanup. Generate a read-only candidate report:
|
|
76
|
+
|
|
77
|
+
```bash
|
|
78
|
+
python3 tools/clis/maggie_deployment.py --retention-plan \
|
|
79
|
+
--release-root /var/www/example \
|
|
80
|
+
--current-link /var/www/example/current \
|
|
81
|
+
--keep-releases 2 --output .maggie/deployment/retention-plan.json
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
The CLI never deletes release directories and never permits fewer than two
|
|
85
|
+
retained releases.
|
|
86
|
+
|
|
73
87
|
Validate the database release separately before a VPS migration:
|
|
74
88
|
|
|
75
89
|
```bash
|
|
@@ -81,6 +95,14 @@ The manifest must reference `DATABASE_URL` (not a literal URL), declare an
|
|
|
81
95
|
advancing version, forward-only/idempotent policy, backup and restore commands,
|
|
82
96
|
and a tested restore artifact. The validator never runs those commands.
|
|
83
97
|
|
|
98
|
+
If a release depends on existing rows or seeded data, set
|
|
99
|
+
`dataDependencies: true` in `.maggie/migration-manifest.json` and provide
|
|
100
|
+
`.maggie/deployment/data-release.json` before deployment. The checkpoint must
|
|
101
|
+
use `maggie-data-release.v1`, declare `requiredTables`, compare
|
|
102
|
+
`previousCounts` and `currentCounts`, and include the release, check time, and
|
|
103
|
+
`passed: true`. Deployment preflight fails closed when this evidence is absent
|
|
104
|
+
or incomplete; it does not copy or mutate data.
|
|
105
|
+
|
|
84
106
|
Validate and materialise scheduled jobs only after their quota and lock policy
|
|
85
107
|
has been reviewed:
|
|
86
108
|
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
name: maggie-design
|
|
3
3
|
description: Design authorized interior pages, review rendered responsive layouts, or explicitly rebrand a packaged homepage/template. Use rebrand only with a named source brand and target brand.
|
|
4
4
|
metadata:
|
|
5
|
-
version: 1.
|
|
5
|
+
version: 1.4.0
|
|
6
6
|
---
|
|
7
7
|
|
|
8
8
|
# Maggie Design
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
name: maggie-seo-geo
|
|
3
3
|
description: Plan, audit, create, rewrite, and measure content for AI CMO's paid SEO and GEO workflow. Use for topic opportunities, AI visibility, technical SEO, extractable article structure, sitemap-based rewrites, GSC readback, or SEO/GEO client reports.
|
|
4
4
|
metadata:
|
|
5
|
-
version: 1.
|
|
5
|
+
version: 1.1.0
|
|
6
6
|
---
|
|
7
7
|
|
|
8
8
|
# Maggie SEO and GEO
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
name: maggie-service-booking
|
|
3
3
|
description: Import, synchronise, validate, and design SPA service pages from a booking provider such as Fresha. Use for service catalogues, treatment variants, prices, durations, booking links, payment links, and booking-aware page generation.
|
|
4
4
|
metadata:
|
|
5
|
-
version: 1.
|
|
5
|
+
version: 1.1.0
|
|
6
6
|
---
|
|
7
7
|
|
|
8
8
|
# Maggie Service Booking
|
|
@@ -5,6 +5,7 @@ from __future__ import annotations
|
|
|
5
5
|
import argparse
|
|
6
6
|
import json
|
|
7
7
|
import re
|
|
8
|
+
from datetime import datetime, timezone
|
|
8
9
|
from pathlib import Path
|
|
9
10
|
|
|
10
11
|
|
|
@@ -57,10 +58,21 @@ def preflight(project: Path, target: str, environment: str) -> dict:
|
|
|
57
58
|
rollback_passed = bool(evidence.get("passed") is True and evidence.get("environment") == "production" and evidence.get("previousRelease") and evidence.get("testedAt"))
|
|
58
59
|
except (OSError, json.JSONDecodeError):
|
|
59
60
|
rollback_passed = False
|
|
60
|
-
|
|
61
|
+
migration_manifest = project / ".maggie" / "migration-manifest.json"
|
|
62
|
+
migration = {}
|
|
63
|
+
if migration_manifest.exists():
|
|
64
|
+
try:
|
|
65
|
+
migration = json.loads(migration_manifest.read_text(encoding="utf-8"))
|
|
66
|
+
except (OSError, json.JSONDecodeError):
|
|
67
|
+
migration = {}
|
|
68
|
+
data_path = project / ".maggie" / "deployment" / "data-release.json"
|
|
69
|
+
data_required = bool(migration.get("dataDependencies"))
|
|
70
|
+
data_checkpoint = validate_data_checkpoint(data_path) if data_path.exists() else {"passed": not data_required, "declared": False, "errors": ["data-release.json is required for a data-dependent release"] if data_required else []}
|
|
71
|
+
retention_ok = bool(plan.get("retention", {}).get("keep") in (2, 3, 4, 5) and plan.get("retention", {}).get("prune") and plan.get("retention", {}).get("preserve"))
|
|
72
|
+
result["checks"].update({"node_runtime_declared": bool(package.get("engines", {}).get("node") or package.get("dependencies", {}).get("astro")), "start_command_or_runtime": "start" in scripts or "preview" in scripts, "stateless_or_migration_plan": True, "release_layout_declared": True, "rollback_command_declared": True, "release_retention_policy": retention_ok, "data_release_checkpoint": data_checkpoint["passed"], "remote_verification_pending": True, "release_plan_artifact": bool(plan_path and plan.get("release_layout") and plan.get("current_release")), "systemd_artifact": bool(systemd_text and "ExecStart=" in systemd_text and "EnvironmentFile=" in systemd_text and "NoNewPrivileges=true" in systemd_text), "nginx_artifact": bool(nginx_text and "proxy_pass" in nginx_text and "server_name" in nginx_text), "deployment_artifacts_secret_free": secret_free, "rollback_evidence": rollback_passed if environment == "production" else True})
|
|
61
73
|
result["vps"] = {"runtime":"Node.js standalone","reverseProxy":"Nginx","processManager":"systemd","releaseLayout":"/var/www/<site>/releases/<release> + current symlink","dns":"Cloudflare DNS or registrar DNS; verify apex and www separately","remoteCommands":["npm ci","npm run build","systemctl restart <service>","systemctl is-active <service>","curl -fsS https://<domain>/robots.txt","curl -fsS https://<domain>/sitemap.xml"],"handoverFields":["host","sshUser","domain","service","release","previousRelease","nginxConfig","migrationVersion","verification","rollback"]}
|
|
62
74
|
result["next_action"] = "review VPS host, SSH user, domain/DNS, systemd service, Nginx config, release path, migrations, and rollback owner before execute"
|
|
63
|
-
result["blocking_checks"] = ["package_manifest", "build_command", "env_example", "node_runtime_declared", "start_command_or_runtime", "stateless_or_migration_plan", "release_layout_declared", "rollback_command_declared", "release_plan_artifact", "systemd_artifact", "nginx_artifact", "deployment_artifacts_secret_free", "rollback_evidence"]
|
|
75
|
+
result["blocking_checks"] = ["package_manifest", "build_command", "env_example", "node_runtime_declared", "start_command_or_runtime", "stateless_or_migration_plan", "release_layout_declared", "rollback_command_declared", "release_retention_policy", "data_release_checkpoint", "release_plan_artifact", "systemd_artifact", "nginx_artifact", "deployment_artifacts_secret_free", "rollback_evidence"]
|
|
64
76
|
else:
|
|
65
77
|
result["blocking_checks"] = ["package_manifest", "build_command", "env_example"]
|
|
66
78
|
return result
|
|
@@ -85,6 +97,7 @@ def vps_plan(domain: str, service: str, release_root: str, node_port: int) -> di
|
|
|
85
97
|
"current_release": current,
|
|
86
98
|
"node_port": node_port,
|
|
87
99
|
"release_layout": f"{root}/releases/<immutable-release> plus current symlink",
|
|
100
|
+
"retention": {"keep": 2, "preserve": [current, f"{root}/releases/<previous-release>"], "prune": f"find {root}/releases -mindepth 1 -maxdepth 1 -type d -printf '%T@ %p\\n' | sort -nr | tail -n +3 | cut -d' ' -f2- | xargs -r rm -rf --"},
|
|
88
101
|
"staging_first": True,
|
|
89
102
|
"secrets": "external environment file; no secrets in generated artifacts",
|
|
90
103
|
"commands": {
|
|
@@ -101,6 +114,38 @@ def vps_plan(domain: str, service: str, release_root: str, node_port: int) -> di
|
|
|
101
114
|
}
|
|
102
115
|
|
|
103
116
|
|
|
117
|
+
def validate_data_checkpoint(path: Path) -> dict:
|
|
118
|
+
try:
|
|
119
|
+
value = json.loads(path.read_text(encoding="utf-8"))
|
|
120
|
+
except (OSError, json.JSONDecodeError) as exc:
|
|
121
|
+
return {"passed": False, "declared": True, "errors": [f"data checkpoint unreadable: {exc}"]}
|
|
122
|
+
required = value.get("requiredTables")
|
|
123
|
+
previous = value.get("previousCounts")
|
|
124
|
+
current = value.get("currentCounts")
|
|
125
|
+
errors = []
|
|
126
|
+
if value.get("schemaVersion") != "maggie-data-release.v1": errors.append("data checkpoint schemaVersion is unsupported")
|
|
127
|
+
if value.get("passed") is not True: errors.append("data checkpoint is not passed")
|
|
128
|
+
if not isinstance(required, list) or not required: errors.append("requiredTables must be a non-empty list")
|
|
129
|
+
if not isinstance(previous, dict) or not isinstance(current, dict): errors.append("previousCounts and currentCounts are required")
|
|
130
|
+
if isinstance(required, list) and isinstance(current, dict):
|
|
131
|
+
missing = [table for table in required if table not in current]
|
|
132
|
+
if missing: errors.append("currentCounts is missing required tables")
|
|
133
|
+
if not value.get("release") or not value.get("checkedAt"): errors.append("release and checkedAt are required")
|
|
134
|
+
return {"passed": not errors, "declared": True, "errors": errors}
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def retention_plan(release_root: str, current_link: str, keep: int) -> dict:
|
|
138
|
+
if keep < 2 or keep > 5:
|
|
139
|
+
raise ValueError("keep must be between 2 and 5")
|
|
140
|
+
root = Path(release_root).resolve()
|
|
141
|
+
releases = sorted((path for path in root.iterdir() if path.is_dir() and not path.is_symlink()), key=lambda path: path.stat().st_mtime, reverse=True) if root.is_dir() else []
|
|
142
|
+
current_target = Path(current_link).resolve() if Path(current_link).exists() else None
|
|
143
|
+
preserved = {str(current_target)} if current_target else set()
|
|
144
|
+
preserved.update(str(path) for path in releases[:keep])
|
|
145
|
+
candidates = [str(path) for path in releases if str(path) not in preserved]
|
|
146
|
+
return {"schemaVersion": "maggie-retention-plan.v1", "releaseRoot": str(root), "currentLink": str(Path(current_link)), "keep": keep, "preserved": sorted(preserved), "candidates": candidates, "mutation": "not executed", "generatedAt": datetime.now(timezone.utc).isoformat()}
|
|
147
|
+
|
|
148
|
+
|
|
104
149
|
def write_vps_plan(output_dir: Path, plan: dict) -> None:
|
|
105
150
|
output_dir.mkdir(parents=True, exist_ok=True)
|
|
106
151
|
domain = plan["domain"]
|
|
@@ -158,6 +203,9 @@ def main() -> int:
|
|
|
158
203
|
parser.add_argument("--release-root", default="/var/www/maggie-site", help="immutable release root")
|
|
159
204
|
parser.add_argument("--node-port", type=int, default=4321)
|
|
160
205
|
parser.add_argument("--plan-dir", help="directory for generated VPS artifacts")
|
|
206
|
+
parser.add_argument("--retention-plan", action="store_true", help="create a read-only release prune candidate plan")
|
|
207
|
+
parser.add_argument("--current-link", help="current symlink for --retention-plan")
|
|
208
|
+
parser.add_argument("--keep-releases", type=int, default=2)
|
|
161
209
|
args = parser.parse_args()
|
|
162
210
|
try:
|
|
163
211
|
if args.vps_plan:
|
|
@@ -168,6 +216,16 @@ def main() -> int:
|
|
|
168
216
|
write_vps_plan(Path(args.plan_dir).resolve(), plan)
|
|
169
217
|
print(json.dumps(plan, indent=2, ensure_ascii=False))
|
|
170
218
|
return 0
|
|
219
|
+
if args.retention_plan:
|
|
220
|
+
if not args.current_link:
|
|
221
|
+
parser.error("--current-link is required with --retention-plan")
|
|
222
|
+
plan = retention_plan(args.release_root, args.current_link, args.keep_releases)
|
|
223
|
+
output = Path(args.output).resolve() if args.output else None
|
|
224
|
+
if output:
|
|
225
|
+
output.parent.mkdir(parents=True, exist_ok=True)
|
|
226
|
+
output.write_text(json.dumps(plan, indent=2) + "\n", encoding="utf-8")
|
|
227
|
+
print(json.dumps(plan, indent=2, ensure_ascii=False))
|
|
228
|
+
return 0
|
|
171
229
|
result = preflight(Path(args.project).resolve(), args.target, args.environment)
|
|
172
230
|
if args.output:
|
|
173
231
|
Path(args.output).parent.mkdir(parents=True, exist_ok=True)
|
|
@@ -526,6 +526,18 @@ def in_place_job(project: Path, routes: list[str], force: bool = False) -> int:
|
|
|
526
526
|
project / "src" / "app" / relative / "page.tsx",
|
|
527
527
|
]
|
|
528
528
|
match = next((path for path in candidates if path.is_file()), None)
|
|
529
|
+
if not match:
|
|
530
|
+
# Dynamic catch-all routes are the source of truth for many page
|
|
531
|
+
# families; a concrete URL still needs to resolve to that route
|
|
532
|
+
# before an in-place design job can be created.
|
|
533
|
+
roots = [project / "src" / "pages", project / "src" / "app"]
|
|
534
|
+
catchalls = {"[...slug]", "[[...slug]]"}
|
|
535
|
+
for root in roots:
|
|
536
|
+
if not root.is_dir():
|
|
537
|
+
continue
|
|
538
|
+
match = next((path for path in root.rglob("*") if path.is_file() and path.stem in catchalls), None)
|
|
539
|
+
if match:
|
|
540
|
+
break
|
|
529
541
|
if not match:
|
|
530
542
|
raise ValueError(f"existing route required for in-place redesign: {route}")
|
|
531
543
|
resolved_routes.append({"route": value, "source": str(match)})
|
|
@@ -186,6 +186,12 @@ def plan_job(args: argparse.Namespace) -> int:
|
|
|
186
186
|
"targetLanguage": args.target_lang,
|
|
187
187
|
"targetLocale": args.locale,
|
|
188
188
|
"operation": args.operation,
|
|
189
|
+
"generationContract": {
|
|
190
|
+
"mode": args.operation,
|
|
191
|
+
"preserveMeaning": args.operation in {"translate", "polish", "localise", "rebrand"},
|
|
192
|
+
"allowStructuralRewrite": args.operation in {"rewrite", "rebrand"},
|
|
193
|
+
"marketAdaptation": args.operation in {"localise", "rebrand"},
|
|
194
|
+
},
|
|
189
195
|
"translationGroupId": identity.get("translationGroupId", f"tg-{content_id}"),
|
|
190
196
|
"sourceRevision": args.source_revision or (source_artifact or {}).get("sourceRevision") or content.get("sourceRevision", "unknown"),
|
|
191
197
|
"status": "draft",
|
|
@@ -218,6 +224,11 @@ def validate_job(path: Path, source_path: Path | None = None, render_path: Path
|
|
|
218
224
|
if not valid_locale(job.get("targetLocale")) or parse_locale(job.get("targetLocale"))[0] != job.get("targetLanguage"):
|
|
219
225
|
errors.append("targetLocale must be a supported tag matching targetLanguage")
|
|
220
226
|
translation = job.get("translation", {})
|
|
227
|
+
generation = job.get("generationContract", {})
|
|
228
|
+
if generation.get("mode") and generation.get("mode") != job.get("operation"):
|
|
229
|
+
errors.append("generationContract mode must match operation")
|
|
230
|
+
if job.get("operation") == "rewrite" and not generation.get("allowStructuralRewrite"):
|
|
231
|
+
errors.append("rewrite requires structural rewrite permission")
|
|
221
232
|
errors.extend(validate_translation({
|
|
222
233
|
"contentId": job.get("contentId"), "lang": job.get("targetLanguage"),
|
|
223
234
|
"locale": job.get("targetLocale"), "translationGroupId": job.get("translationGroupId", f"tg-{job.get('contentId')}"),
|
|
@@ -793,9 +793,9 @@ def cmd_match_pages_review(args):
|
|
|
793
793
|
out=project/".maggie"/"booking"/"page-matches.json"; out.parent.mkdir(parents=True,exist_ok=True); (project/"docs").mkdir(parents=True,exist_ok=True); payload={"matchedAt":NOW(),"requiresManualSelection":True,"candidateServices":report,"selected":0,"conflicts":conflicts}; out.write_text(json.dumps(payload,indent=2,ensure_ascii=False)+"\n",encoding="utf-8"); (project/"docs"/"service-page-matches.json").write_text(json.dumps(payload,indent=2,ensure_ascii=False)+"\n",encoding="utf-8"); print(json.dumps({"status":"conflict-review","pagesScanned":len(candidates),"servicesScanned":len(report),"selected":0,"conflicts":conflicts,"report":str(out)},indent=2)); return 1
|
|
794
794
|
for service,path_value in selected:
|
|
795
795
|
relation={"path":path_value,"role":"canonical","matchMethod":"manual-selection","confidence":1.0}; service["pages"]=[p for p in service.get("pages",[]) if p.get("path")!=path_value]+[relation]
|
|
796
|
-
|
|
797
|
-
|
|
798
|
-
|
|
796
|
+
# Reviewed selections must not erase existing supporting relations. They
|
|
797
|
+
# may have been imported from a route table or authored intentionally even
|
|
798
|
+
# when the current page scanner cannot see their source file.
|
|
799
799
|
out=project/".maggie"/"booking"/"page-matches.json"; payload={"matchedAt":NOW(),"requiresManualSelection":True,"candidateServices":report,"selected":len(selected),"conflicts":[]}; out.parent.mkdir(parents=True,exist_ok=True); (project/"docs").mkdir(parents=True,exist_ok=True); out.write_text(json.dumps(payload,indent=2,ensure_ascii=False)+"\n",encoding="utf-8"); (project/"docs"/"service-page-matches.json").write_text(json.dumps(payload,indent=2,ensure_ascii=False)+"\n",encoding="utf-8"); save(project,data); print(json.dumps({"status":"review-required","pagesScanned":len(candidates),"servicesScanned":len(report),"selected":len(selected),"conflicts":[],"report":str(out)},indent=2)); return 0
|
|
800
800
|
def cmd_run(args):
|
|
801
801
|
project=root(args); job={"workflow":"maggie-service-booking","phase":"created","source":args.source,"provider":args.provider,"startedAt":NOW(),"history":[]}
|
|
@@ -32,6 +32,7 @@ class PageParser(HTMLParser):
|
|
|
32
32
|
self._jsonld = None
|
|
33
33
|
self.images = []
|
|
34
34
|
self.hreflang = []
|
|
35
|
+
self.robots_directives = []
|
|
35
36
|
|
|
36
37
|
def handle_starttag(self, tag, attrs):
|
|
37
38
|
data = dict(attrs)
|
|
@@ -39,6 +40,8 @@ class PageParser(HTMLParser):
|
|
|
39
40
|
self.lang = data.get("lang", "")
|
|
40
41
|
if tag == "meta" and data.get("name"):
|
|
41
42
|
self.meta[data["name"].lower()] = data.get("content", "")
|
|
43
|
+
if data["name"].lower() == "robots":
|
|
44
|
+
self.robots_directives.append(data.get("content", "").lower())
|
|
42
45
|
if tag == "meta" and data.get("property"):
|
|
43
46
|
self.meta[data["property"].lower()] = data.get("content", "")
|
|
44
47
|
if tag == "title":
|
|
@@ -86,6 +89,8 @@ def audit_page(url: str, html: str, status: int, content_type: str, expected_lan
|
|
|
86
89
|
page = PageParser()
|
|
87
90
|
page.feed(html)
|
|
88
91
|
canonical = urljoin(url, page.canonical) if page.canonical else ""
|
|
92
|
+
robots = page.robots_directives
|
|
93
|
+
robots_tokens = [token.strip() for directive in robots for token in directive.split(",")]
|
|
89
94
|
return {
|
|
90
95
|
"url": url,
|
|
91
96
|
"status": status,
|
|
@@ -105,6 +110,9 @@ def audit_page(url: str, html: str, status: int, content_type: str, expected_lan
|
|
|
105
110
|
"image_count": len(page.images),
|
|
106
111
|
"language": parse_locale(page.lang)[0],
|
|
107
112
|
"hreflang": page.hreflang,
|
|
113
|
+
"robots": not robots or not any(token in {"noindex", "none", "nofollow"} for token in robots_tokens),
|
|
114
|
+
"robots_directives": robots,
|
|
115
|
+
"robots_conflict": len({token for token in robots_tokens if token in {"index", "noindex", "follow", "nofollow", "none"}} & {"index", "noindex"}) > 1 or len({token for token in robots_tokens if token in {"follow", "nofollow", "none"}} & {"follow", "nofollow"}) > 1,
|
|
108
116
|
}
|
|
109
117
|
|
|
110
118
|
|
|
@@ -197,6 +205,9 @@ def main() -> int:
|
|
|
197
205
|
checks["jsonld"] = {"ok": page.jsonld > 0 and all(item is not None for item in page.jsonld_values), "count": page.jsonld}
|
|
198
206
|
checks["entity_jsonld"] = {"ok": any(isinstance(item, dict) and item.get("@type") and (item.get("url") or item.get("@id")) for item in page.jsonld_values), "count": page.jsonld}
|
|
199
207
|
checks["crawlable_links"] = {"ok": page.anchors > 0, "count": page.anchors}
|
|
208
|
+
robots_tokens = [token.strip() for directive in page.robots_directives for token in directive.split(",")]
|
|
209
|
+
checks["robots_directive"] = {"ok": not page.robots_directives or not any(token in {"noindex", "none", "nofollow"} for token in robots_tokens), "directives": page.robots_directives}
|
|
210
|
+
checks["robots_conflict"] = {"ok": not (len({token for token in robots_tokens if token in {"index", "noindex"}}) > 1 or len({token for token in robots_tokens if token in {"follow", "nofollow"}}) > 1), "directives": page.robots_directives}
|
|
200
211
|
if args.check_hreflang or args.check_translation_completeness:
|
|
201
212
|
checks["hreflang"] = hreflang_check(base, page, expected_languages) if args.check_hreflang else {"ok": True, "links": page.hreflang}
|
|
202
213
|
if args.check_translation_completeness and expected_languages:
|
|
@@ -233,7 +244,7 @@ def main() -> int:
|
|
|
233
244
|
page_checks["translation_completeness"] = {"ok": all(lang in {item["lang"] for item in parsed.hreflang} for lang in expected_languages), "expected": sorted(expected_languages)}
|
|
234
245
|
page_checks["passed"] = all(
|
|
235
246
|
page_checks[key]
|
|
236
|
-
for key in ("ok", "title", "h1", "canonical", "description", "locale", "open_graph", "twitter_card", "jsonld", "entity_jsonld", "images")
|
|
247
|
+
for key in ("ok", "title", "h1", "canonical", "description", "locale", "open_graph", "twitter_card", "jsonld", "entity_jsonld", "images", "robots", "robots_conflict")
|
|
237
248
|
) and (not args.check_hreflang or page_checks["hreflang_check"]["ok"]) and (not args.check_translation_completeness or page_checks.get("translation_completeness", {}).get("ok", False))
|
|
238
249
|
except Exception as exc:
|
|
239
250
|
page_checks = {"url": page_url, "passed": False, "error": type(exc).__name__}
|
|
@@ -3,6 +3,7 @@
|
|
|
3
3
|
from __future__ import annotations
|
|
4
4
|
|
|
5
5
|
import hashlib
|
|
6
|
+
import json
|
|
6
7
|
import re
|
|
7
8
|
from datetime import datetime, timezone
|
|
8
9
|
from html import escape
|
|
@@ -27,10 +28,18 @@ def parse_routes(text: str) -> list[dict[str, str]]:
|
|
|
27
28
|
continue
|
|
28
29
|
parts = line.split("\t")
|
|
29
30
|
if len(parts) < 2:
|
|
30
|
-
raise ValueError("route lines must be type<TAB>absolute-url<TAB>optional-lastmod")
|
|
31
|
+
raise ValueError("route lines must be type<TAB>absolute-url<TAB>optional-lastmod<TAB>optional-alternates-json")
|
|
31
32
|
item = {"type": parts[0].strip(), "url": parts[1].strip()}
|
|
32
33
|
if len(parts) > 2 and parts[2].strip():
|
|
33
34
|
item["lastmod"] = parts[2].strip()
|
|
35
|
+
if len(parts) > 3 and parts[3].strip():
|
|
36
|
+
try:
|
|
37
|
+
alternates = json.loads(parts[3])
|
|
38
|
+
except json.JSONDecodeError as exc:
|
|
39
|
+
raise ValueError("route alternates must be valid JSON") from exc
|
|
40
|
+
if not isinstance(alternates, dict):
|
|
41
|
+
raise ValueError("route alternates must be a JSON object")
|
|
42
|
+
item["alternates"] = alternates
|
|
34
43
|
routes.append(item)
|
|
35
44
|
return routes
|
|
36
45
|
|
|
@@ -63,8 +72,9 @@ def xml_file(urls: list[dict[str, str]]) -> str:
|
|
|
63
72
|
rows = []
|
|
64
73
|
for item in urls:
|
|
65
74
|
lastmod = f"<lastmod>{escape(item['lastmod'])}</lastmod>" if item.get("lastmod") else ""
|
|
66
|
-
|
|
67
|
-
|
|
75
|
+
alternate_xml = "".join(f'<xhtml:link rel="alternate" hreflang="{escape(str(lang))}" href="{escape(str(url))}" />' for lang, url in sorted((item.get("alternates") or {}).items()))
|
|
76
|
+
rows.append(f"<url><loc>{escape(item['url'])}</loc>{lastmod}{alternate_xml}</url>")
|
|
77
|
+
return '<?xml version="1.0" encoding="UTF-8"?>\n<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9" xmlns:xhtml="http://www.w3.org/1999/xhtml">' + "".join(rows) + "</urlset>\n"
|
|
68
78
|
|
|
69
79
|
|
|
70
80
|
def build_plan(routes: list[dict[str, str]], origin: str, content_types: set[str], chunk_target: int = DEFAULT_CHUNK_TARGET, previous: dict | None = None) -> dict:
|
|
@@ -112,6 +122,9 @@ def validate_plan_data(plan: dict) -> dict:
|
|
|
112
122
|
locs = [element.text or "" for element in root.iter() if element.tag.endswith("loc")]
|
|
113
123
|
if any(not absolute_url(value, origin) for value in locs):
|
|
114
124
|
errors.append(f"chunk contains a relative or off-origin loc: {chunk.get('filename')}")
|
|
125
|
+
for link in root.iter():
|
|
126
|
+
if link.tag.endswith("link") and link.get("rel") == "alternate" and not absolute_url(link.get("href", ""), origin):
|
|
127
|
+
errors.append(f"chunk contains a relative or off-origin alternate: {chunk.get('filename')}")
|
|
115
128
|
except ElementTree.ParseError:
|
|
116
129
|
errors.append(f"chunk is not valid XML: {chunk.get('filename')}")
|
|
117
130
|
index = plan.get("index", {})
|