@topy-ai/maggie 0.7.16 → 0.7.18
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +52 -9
- package/README.zh-TW.md +10 -2
- package/bin/maggie.js +5 -4
- package/bundled-references/universal-booking-adapter.md +36 -1
- package/bundled-skills/maggie-blog/SKILL.md +11 -1
- package/bundled-skills/maggie-dash/SKILL.md +5 -0
- package/bundled-skills/maggie-deployment/SKILL.md +9 -1
- package/bundled-skills/maggie-feedback/SKILL.md +5 -0
- package/bundled-skills/maggie-ops/SKILL.md +8 -0
- package/bundled-skills/maggie-qa-workflow/SKILL.md +11 -0
- package/bundled-skills/maggie-seo-geo/SKILL.md +29 -1
- package/bundled-skills/maggie-service-booking/SKILL.md +19 -1
- package/bundled-tools/clis/maggie.py +1 -1
- package/bundled-tools/clis/maggie_blog.py +12 -2
- package/bundled-tools/clis/maggie_feedback.py +17 -5
- package/bundled-tools/clis/maggie_head_tags.py +35 -0
- package/bundled-tools/clis/maggie_indexnow.py +6 -3
- package/bundled-tools/clis/maggie_ops.py +12 -0
- package/bundled-tools/clis/maggie_qa_workflow.py +2 -0
- package/bundled-tools/clis/maggie_release.py +88 -2
- package/bundled-tools/clis/maggie_service_booking.py +42 -3
- package/bundled-tools/clis/maggie_social_cards.py +45 -0
- package/bundled-tools/clis/site_audit.py +2 -2
- package/bundled-tools/runtime/booking_capabilities.py +137 -0
- package/bundled-tools/runtime/maggie_blog.py +85 -0
- package/bundled-tools/runtime/maggie_favicon.py +86 -0
- package/bundled-tools/runtime/maggie_head_tags.py +101 -0
- package/bundled-tools/runtime/maggie_indexnow.py +50 -0
- package/bundled-tools/runtime/maggie_social_cards.py +165 -0
- package/bundled-tools/runtime/service_variants.py +48 -7
- package/package.json +1 -1
- package/references/universal-booking-adapter.md +36 -1
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
"""Audit normalized head metadata across rendered shells."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
from collections import defaultdict
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from urllib.parse import urlparse
|
|
9
|
+
from typing import Any
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
SCHEMA = "maggie-head-tags.v1"
|
|
13
|
+
STABLE_TAGS = ("icon", "og:image", "og:image:type", "verification", "robots", "twitter:card")
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def _extension(value: object) -> str | None:
|
|
17
|
+
path = urlparse(str(value or "")).path.lower()
|
|
18
|
+
if "." not in path.rsplit("/", 1)[-1]:
|
|
19
|
+
return None
|
|
20
|
+
return "." + path.rsplit("/", 1)[-1].rsplit(".", 1)[-1]
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _signature(name: str, value: object, tags: dict[str, Any]) -> dict[str, Any]:
|
|
24
|
+
"""Return only safe, structural metadata; never echo verification values."""
|
|
25
|
+
if name == "verification":
|
|
26
|
+
present = bool(value) if not isinstance(value, dict) else bool(value.get("present", value.get("content")))
|
|
27
|
+
return {"present": present}
|
|
28
|
+
if name == "icon":
|
|
29
|
+
data = value if isinstance(value, dict) else {"href": value}
|
|
30
|
+
return {"present": bool(value), "rel": str(data.get("rel") or "icon"), "type": str(data.get("type") or ""), "extension": _extension(data.get("href"))}
|
|
31
|
+
if name == "og:image":
|
|
32
|
+
data = value if isinstance(value, dict) else {"href": value}
|
|
33
|
+
image_type = data.get("type") or tags.get("og:image:type") or ""
|
|
34
|
+
return {"present": bool(value), "type": str(image_type), "extension": _extension(data.get("href") or data.get("content") or value), "fallback": bool(data.get("fallback", tags.get("og:image:fallback", False)))}
|
|
35
|
+
if name == "og:image:type":
|
|
36
|
+
return {"present": bool(value), "type": str(value or "")}
|
|
37
|
+
if isinstance(value, dict):
|
|
38
|
+
return {"present": bool(value), "type": str(value.get("type") or "")}
|
|
39
|
+
return {"present": bool(value), "value": str(value or "")}
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def audit_head_tags(pages: list[dict[str, Any]]) -> dict[str, Any]:
|
|
43
|
+
errors: list[dict[str, str]] = []
|
|
44
|
+
observations: list[dict[str, Any]] = []
|
|
45
|
+
for page in pages:
|
|
46
|
+
if not isinstance(page, dict) or not str(page.get("url") or "").strip():
|
|
47
|
+
errors.append({"reason": "each page requires a URL"})
|
|
48
|
+
continue
|
|
49
|
+
shell = str(page.get("shell") or "").strip()
|
|
50
|
+
if not shell:
|
|
51
|
+
errors.append({"url": str(page["url"]), "reason": "each page requires a shell"})
|
|
52
|
+
continue
|
|
53
|
+
tags = page.get("tags", {})
|
|
54
|
+
if not isinstance(tags, dict):
|
|
55
|
+
errors.append({"url": str(page["url"]), "reason": "tags must be an object"})
|
|
56
|
+
continue
|
|
57
|
+
observations.append({"url": str(page["url"]), "shell": shell, "tags": {name: _signature(name, tags.get(name), tags) for name in STABLE_TAGS}})
|
|
58
|
+
|
|
59
|
+
by_shell: dict[str, list[dict[str, Any]]] = defaultdict(list)
|
|
60
|
+
for observation in observations:
|
|
61
|
+
by_shell[observation["shell"]].append(observation)
|
|
62
|
+
disagreements: list[dict[str, Any]] = []
|
|
63
|
+
|
|
64
|
+
def compare(scope: str, label: str, entries: list[tuple[str, dict[str, Any]]]) -> None:
|
|
65
|
+
for tag in STABLE_TAGS:
|
|
66
|
+
variants: dict[str, dict[str, Any]] = {}
|
|
67
|
+
urls: dict[str, list[str]] = defaultdict(list)
|
|
68
|
+
for identifier, observation in entries:
|
|
69
|
+
signature = observation["tags"][tag]
|
|
70
|
+
key = json.dumps(signature, sort_keys=True, separators=(",", ":"))
|
|
71
|
+
variants[key] = signature
|
|
72
|
+
urls[key].append(observation["url"])
|
|
73
|
+
if len(variants) > 1:
|
|
74
|
+
disagreements.append({"scope": scope, "shell": label if scope == "shell" else None, "tag": tag, "variants": [{"signature": signature, "urls": sorted(urls[key])} for key, signature in sorted(variants.items())]})
|
|
75
|
+
|
|
76
|
+
for shell, shell_pages in sorted(by_shell.items()):
|
|
77
|
+
compare("shell", shell, [(shell, page) for page in shell_pages])
|
|
78
|
+
for tag in STABLE_TAGS:
|
|
79
|
+
entries = [(shell, page) for shell, shell_pages in sorted(by_shell.items()) for page in shell_pages]
|
|
80
|
+
shell_signatures: dict[str, dict[str, Any]] = {}
|
|
81
|
+
shell_urls: dict[str, list[str]] = defaultdict(list)
|
|
82
|
+
for shell, page in entries:
|
|
83
|
+
key = json.dumps(page["tags"][tag], sort_keys=True, separators=(",", ":"))
|
|
84
|
+
shell_signatures[shell] = page["tags"][tag]
|
|
85
|
+
shell_urls[key].append(page["url"])
|
|
86
|
+
unique = {json.dumps(value, sort_keys=True, separators=(",", ":")) for value in shell_signatures.values()}
|
|
87
|
+
if len(unique) > 1:
|
|
88
|
+
disagreements.append({"scope": "across-shells", "shell": None, "tag": tag, "variants": [{"signature": shell_signatures[shell], "shell": shell, "urls": sorted(shell_urls[json.dumps(shell_signatures[shell], sort_keys=True, separators=(",", ":"))])} for shell in sorted(shell_signatures)]})
|
|
89
|
+
|
|
90
|
+
coverage = []
|
|
91
|
+
for shell, shell_pages in sorted(by_shell.items()):
|
|
92
|
+
coverage.append({"shell": shell, "pageCount": len(shell_pages), "tags": {tag: sum(1 for page in shell_pages if page["tags"][tag]["present"]) for tag in STABLE_TAGS}})
|
|
93
|
+
return {"schemaVersion": SCHEMA, "shells": coverage, "observations": observations, "disagreements": disagreements, "errors": errors, "passed": not errors and not disagreements, "mutation": False}
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def read_pages(path: Path) -> list[dict[str, Any]]:
|
|
97
|
+
value = json.loads(path.read_text(encoding="utf-8"))
|
|
98
|
+
pages = value.get("pages") if isinstance(value, dict) else value
|
|
99
|
+
if not isinstance(pages, list):
|
|
100
|
+
raise ValueError("pages file must be a JSON array or {\"pages\": []}")
|
|
101
|
+
return pages
|
|
@@ -99,6 +99,56 @@ def check_key_file(public_dir: Path, key_file: str, key: str) -> dict[str, Any]:
|
|
|
99
99
|
"matches": served == key.strip(), "status": "pass" if served == key.strip() else "fail"}
|
|
100
100
|
|
|
101
101
|
|
|
102
|
+
def _public_url(value: str, label: str) -> str:
|
|
103
|
+
parsed = urlparse(value)
|
|
104
|
+
if parsed.scheme not in {"http", "https"} or not parsed.netloc or parsed.fragment:
|
|
105
|
+
raise ValueError(f"{label} must be an absolute HTTP(S) URL without a fragment")
|
|
106
|
+
return value
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def _fetch_key(url: str, expected: str, timeout: int) -> dict[str, Any]:
|
|
110
|
+
status_code: int | None = None
|
|
111
|
+
matches = False
|
|
112
|
+
try:
|
|
113
|
+
request = Request(url, headers={"User-Agent": "Maggie-IndexNow-Key-Check/1"})
|
|
114
|
+
with urlopen(request, timeout=timeout) as response:
|
|
115
|
+
status_code = int(response.status)
|
|
116
|
+
body = response.read(4097)
|
|
117
|
+
matches = len(body) <= 4096 and body.decode("utf-8", "replace").strip() == expected
|
|
118
|
+
except HTTPError as error:
|
|
119
|
+
status_code = int(error.code)
|
|
120
|
+
except (URLError, TimeoutError, OSError):
|
|
121
|
+
pass
|
|
122
|
+
return {"statusCode": status_code, "matches": matches}
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def check_key_http(key_url: str, key: str, wrong_key_url: str | None = None,
|
|
126
|
+
timeout: int = 15) -> dict[str, Any]:
|
|
127
|
+
"""Verify the deployed key route without exposing its response body."""
|
|
128
|
+
expected = key.strip()
|
|
129
|
+
if not expected:
|
|
130
|
+
raise ValueError("key is required")
|
|
131
|
+
if timeout < 1:
|
|
132
|
+
raise ValueError("timeout must be positive")
|
|
133
|
+
correct_url = _public_url(key_url, "key-url")
|
|
134
|
+
correct = _fetch_key(correct_url, expected, timeout)
|
|
135
|
+
result: dict[str, Any] = {
|
|
136
|
+
"keyUrl": correct_url,
|
|
137
|
+
"correct": correct,
|
|
138
|
+
"wrong": None,
|
|
139
|
+
"status": "pass" if correct["statusCode"] == 200 and correct["matches"] else "fail",
|
|
140
|
+
}
|
|
141
|
+
if wrong_key_url:
|
|
142
|
+
wrong_url = _public_url(wrong_key_url, "wrong-key-url")
|
|
143
|
+
if wrong_url == correct_url:
|
|
144
|
+
raise ValueError("wrong-key-url must differ from key-url")
|
|
145
|
+
wrong = _fetch_key(wrong_url, expected, timeout)
|
|
146
|
+
result["wrong"] = {"url": wrong_url, **wrong, "is404": wrong["statusCode"] == 404}
|
|
147
|
+
if not result["wrong"]["is404"]:
|
|
148
|
+
result["status"] = "fail"
|
|
149
|
+
return result
|
|
150
|
+
|
|
151
|
+
|
|
102
152
|
def submit_plan(plan: dict[str, Any], state: dict[str, Any], endpoint: str) -> dict[str, Any]:
|
|
103
153
|
urls = plan.get("eligible") if isinstance(plan.get("eligible"), list) else []
|
|
104
154
|
if not urls:
|
|
@@ -0,0 +1,165 @@
|
|
|
1
|
+
"""Provider-neutral Open Graph social-card inspection."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
from collections import Counter
|
|
7
|
+
from html.parser import HTMLParser
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from typing import Any
|
|
10
|
+
from urllib.error import HTTPError, URLError
|
|
11
|
+
from urllib.parse import urljoin, urlparse
|
|
12
|
+
from urllib.request import Request, urlopen
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
SCHEMA = "maggie-social-cards.v1"
|
|
16
|
+
MIN_WIDTH = 300
|
|
17
|
+
MIN_HEIGHT = 157
|
|
18
|
+
SUPPORTED_FORMATS = {"image/jpeg", "image/png", "image/gif", "image/x-icon"}
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class OpenGraphParser(HTMLParser):
|
|
22
|
+
def __init__(self) -> None:
|
|
23
|
+
super().__init__()
|
|
24
|
+
self.values: dict[str, str] = {}
|
|
25
|
+
|
|
26
|
+
def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
|
|
27
|
+
if tag != "meta":
|
|
28
|
+
return
|
|
29
|
+
data = dict(attrs)
|
|
30
|
+
key = (data.get("property") or data.get("name") or "").lower()
|
|
31
|
+
if key in {"og:image", "og:image:type"} and key not in self.values:
|
|
32
|
+
self.values[key] = str(data.get("content") or "").strip()
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _dimensions(raw: bytes, content_type: str) -> tuple[str | None, int | None, int | None]:
|
|
36
|
+
"""Read dimensions from common headers without an image dependency."""
|
|
37
|
+
if raw.startswith(b"\x89PNG\r\n\x1a\n") and len(raw) >= 24:
|
|
38
|
+
return "image/png", int.from_bytes(raw[16:20], "big"), int.from_bytes(raw[20:24], "big")
|
|
39
|
+
if raw[:3] == b"GIF" and len(raw) >= 10:
|
|
40
|
+
return "image/gif", int.from_bytes(raw[6:8], "little"), int.from_bytes(raw[8:10], "little")
|
|
41
|
+
if raw[:4] == b"\x00\x00\x01\x00" and len(raw) >= 22:
|
|
42
|
+
width = raw[6] or 256
|
|
43
|
+
height = raw[7] or 256
|
|
44
|
+
return "image/x-icon", width, height
|
|
45
|
+
if raw[:2] == b"\xff\xd8":
|
|
46
|
+
index = 2
|
|
47
|
+
while index + 9 < len(raw):
|
|
48
|
+
if raw[index] != 0xFF:
|
|
49
|
+
index += 1
|
|
50
|
+
continue
|
|
51
|
+
marker = raw[index + 1]
|
|
52
|
+
index += 2
|
|
53
|
+
if marker in {0xD8, 0xD9}:
|
|
54
|
+
continue
|
|
55
|
+
if index + 2 > len(raw):
|
|
56
|
+
break
|
|
57
|
+
segment_length = int.from_bytes(raw[index:index + 2], "big")
|
|
58
|
+
if segment_length < 2 or index + segment_length > len(raw):
|
|
59
|
+
break
|
|
60
|
+
if marker in set(range(0xC0, 0xC4)) | set(range(0xC5, 0xC8)) | set(range(0xC9, 0xCC)) | set(range(0xCD, 0xD0)):
|
|
61
|
+
if segment_length >= 7:
|
|
62
|
+
return "image/jpeg", int.from_bytes(raw[index + 5:index + 7], "big"), int.from_bytes(raw[index + 3:index + 5], "big")
|
|
63
|
+
index += segment_length
|
|
64
|
+
return "image/jpeg", None, None
|
|
65
|
+
if raw[:4] == b"RIFF" and raw[8:12] == b"WEBP" and len(raw) >= 30:
|
|
66
|
+
if raw[12:16] == b"VP8X":
|
|
67
|
+
width = 1 + int.from_bytes(raw[24:27], "little")
|
|
68
|
+
height = 1 + int.from_bytes(raw[27:30], "little")
|
|
69
|
+
return "image/webp", width, height
|
|
70
|
+
return "image/webp", None, None
|
|
71
|
+
return (content_type.split(";", 1)[0].lower() or None), None, None
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _fetch(url: str, timeout: int, limit: int = 8_000_000) -> tuple[int | None, str, bytes, str | None]:
|
|
75
|
+
try:
|
|
76
|
+
request = Request(url, headers={"User-Agent": "Maggie-Social-Card-Audit/1"})
|
|
77
|
+
with urlopen(request, timeout=timeout) as response:
|
|
78
|
+
raw = response.read(limit + 1)
|
|
79
|
+
return int(response.status), response.headers.get_content_type(), raw[:limit], None if len(raw) <= limit else "response exceeds size limit"
|
|
80
|
+
except HTTPError as error:
|
|
81
|
+
return int(error.code), "", b"", "HTTP error"
|
|
82
|
+
except (URLError, TimeoutError, OSError):
|
|
83
|
+
return None, "", b"", "request failed"
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def _absolute(value: str, base: str) -> str:
|
|
87
|
+
candidate = urljoin(base.rstrip("/") + "/", value)
|
|
88
|
+
parsed = urlparse(candidate)
|
|
89
|
+
if parsed.scheme not in {"http", "https"} or not parsed.netloc or parsed.fragment:
|
|
90
|
+
raise ValueError("og:image must resolve to an absolute HTTP(S) URL without a fragment")
|
|
91
|
+
return candidate
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def audit_cards(pages: list[dict[str, Any]], *, timeout: int = 15,
|
|
95
|
+
min_width: int = MIN_WIDTH, min_height: int = MIN_HEIGHT) -> dict[str, Any]:
|
|
96
|
+
if timeout < 1 or min_width < 1 or min_height < 1:
|
|
97
|
+
raise ValueError("timeout and image dimensions must be positive")
|
|
98
|
+
reports: list[dict[str, Any]] = []
|
|
99
|
+
errors: list[dict[str, str]] = []
|
|
100
|
+
sources: list[str] = []
|
|
101
|
+
for page in pages:
|
|
102
|
+
if not isinstance(page, dict) or not str(page.get("url") or "").strip():
|
|
103
|
+
errors.append({"reason": "page must contain a URL"})
|
|
104
|
+
continue
|
|
105
|
+
page_url = str(page["url"]).strip()
|
|
106
|
+
image = str(page.get("ogImage") or page.get("og:image") or "").strip()
|
|
107
|
+
record: dict[str, Any] = {"url": page_url, "ogImage": image or None, "fallback": bool(page.get("fallback", False)), "errors": []}
|
|
108
|
+
if not image:
|
|
109
|
+
record["errors"].append("missing og:image")
|
|
110
|
+
reports.append(record)
|
|
111
|
+
errors.append({"url": page_url, "reason": "missing og:image"})
|
|
112
|
+
continue
|
|
113
|
+
try:
|
|
114
|
+
image_url = _absolute(image, page_url)
|
|
115
|
+
except ValueError as exc:
|
|
116
|
+
record["errors"].append(str(exc))
|
|
117
|
+
reports.append(record)
|
|
118
|
+
errors.append({"url": page_url, "reason": str(exc)})
|
|
119
|
+
continue
|
|
120
|
+
status, header_type, raw, fetch_error = _fetch(image_url, timeout)
|
|
121
|
+
image_type, width, height = _dimensions(raw, header_type)
|
|
122
|
+
record.update({"imageUrl": image_url, "statusCode": status, "format": image_type, "width": width, "height": height})
|
|
123
|
+
if fetch_error or status != 200:
|
|
124
|
+
record["errors"].append(fetch_error or "image request did not return HTTP 200")
|
|
125
|
+
elif image_type not in SUPPORTED_FORMATS:
|
|
126
|
+
record["errors"].append("unsupported format for large social card")
|
|
127
|
+
elif width is None or height is None:
|
|
128
|
+
record["errors"].append("image dimensions could not be read")
|
|
129
|
+
else:
|
|
130
|
+
if width < min_width or height < min_height:
|
|
131
|
+
record["errors"].append(f"image is smaller than {min_width}x{min_height}")
|
|
132
|
+
reports.append(record)
|
|
133
|
+
sources.append(image_url)
|
|
134
|
+
for reason in record["errors"]:
|
|
135
|
+
errors.append({"url": page_url, "reason": reason})
|
|
136
|
+
|
|
137
|
+
warnings: list[dict[str, Any]] = []
|
|
138
|
+
if len(sources) >= 3:
|
|
139
|
+
source, count = Counter(sources).most_common(1)[0]
|
|
140
|
+
if count / len(sources) >= 0.75:
|
|
141
|
+
warnings.append({"reason": "generic fallback concentration", "imageUrl": source, "count": count, "sampleSize": len(sources), "ratio": round(count / len(sources), 3)})
|
|
142
|
+
return {"schemaVersion": SCHEMA, "threshold": {"minWidth": min_width, "minHeight": min_height, "supportedFormats": sorted(SUPPORTED_FORMATS)}, "pages": reports, "warnings": warnings, "errors": errors, "passed": not errors, "mutation": False}
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def pages_from_urls(urls: list[str], timeout: int) -> list[dict[str, Any]]:
|
|
146
|
+
pages: list[dict[str, Any]] = []
|
|
147
|
+
for page_url in urls:
|
|
148
|
+
status, content_type, raw, fetch_error = _fetch(page_url, timeout, 2_000_000)
|
|
149
|
+
if fetch_error or status != 200 or content_type != "text/html":
|
|
150
|
+
pages.append({"url": page_url, "ogImage": "", "fetchError": fetch_error or "page request did not return HTML"})
|
|
151
|
+
continue
|
|
152
|
+
parser = OpenGraphParser()
|
|
153
|
+
parser.feed(raw.decode("utf-8", "replace"))
|
|
154
|
+
pages.append({"url": page_url, "ogImage": parser.values.get("og:image", "")})
|
|
155
|
+
return pages
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def read_pages(path: Path) -> list[dict[str, Any]]:
|
|
159
|
+
value = json.loads(path.read_text(encoding="utf-8"))
|
|
160
|
+
pages = value.get("pages") if isinstance(value, dict) else value
|
|
161
|
+
if not isinstance(pages, list):
|
|
162
|
+
raise ValueError("pages file must be a JSON array or {\"pages\": []}")
|
|
163
|
+
if any(not isinstance(page, dict) for page in pages):
|
|
164
|
+
raise ValueError("pages file entries must be objects")
|
|
165
|
+
return pages
|
|
@@ -35,6 +35,16 @@ def variant_slug(kind: str, *, location: str = "", event: str = "", holiday: str
|
|
|
35
35
|
return f"{kind}/{slug_part(value)}"
|
|
36
36
|
|
|
37
37
|
|
|
38
|
+
def locale_segment(value: str) -> str:
|
|
39
|
+
"""Return the stable lowercase URL segment for a BCP 47-like locale."""
|
|
40
|
+
return slug_part(value.replace("_", "-"))
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def variant_canonical_url(slug: str, locale: str, market: str) -> str:
|
|
44
|
+
"""Build a collision-safe, market/language-aware service URL."""
|
|
45
|
+
return f"/services/{slug_part(market)}/{locale_segment(locale)}/{slug.strip('/')}/"
|
|
46
|
+
|
|
47
|
+
|
|
38
48
|
def similarity(source: str, variant: str) -> float:
|
|
39
49
|
left, right = set(re.findall(r"[a-z0-9]+", source.lower())), set(re.findall(r"[a-z0-9]+", variant.lower()))
|
|
40
50
|
return 1.0 if not left and not right else len(left & right) / max(1, len(left | right))
|
|
@@ -62,6 +72,20 @@ class ServiceVariantStore:
|
|
|
62
72
|
raise ValueError("service variant not found")
|
|
63
73
|
return found
|
|
64
74
|
|
|
75
|
+
def _cluster(self, canonical_variant_id: str) -> list[dict]:
|
|
76
|
+
return [
|
|
77
|
+
item for item in self.data["variants"]
|
|
78
|
+
if item.get("id") == canonical_variant_id or item.get("canonicalVariantId") == canonical_variant_id
|
|
79
|
+
]
|
|
80
|
+
|
|
81
|
+
def _refresh_hreflang(self, canonical_variant_id: str) -> None:
|
|
82
|
+
"""Keep published locale links reciprocal and limited to published peers."""
|
|
83
|
+
cluster = self._cluster(canonical_variant_id)
|
|
84
|
+
published = [item for item in cluster if item.get("status") == "published"]
|
|
85
|
+
published_links = {item["locale"]: item["canonicalUrl"] for item in published}
|
|
86
|
+
for item in cluster:
|
|
87
|
+
item["hreflang"] = published_links if item in published else {item["locale"]: item["canonicalUrl"]}
|
|
88
|
+
|
|
65
89
|
def create(self, *, service_id: str, variant_id: str, variant_type: str, locale: str, market: str,
|
|
66
90
|
slug: str, title: str, facts: list[dict], source_revision: str, canonical_variant_id: str | None = None,
|
|
67
91
|
cluster_links: list[str] | None = None, layout_family: str = "service-default", actor: str = "cli") -> dict:
|
|
@@ -73,15 +97,19 @@ class ServiceVariantStore:
|
|
|
73
97
|
raise ValueError("variant requires meaningful keyed facts")
|
|
74
98
|
if any(item["id"] == variant_id for item in self.data["variants"]):
|
|
75
99
|
raise ValueError("variant ID already exists")
|
|
76
|
-
if any(item["slug"] == slug and item["locale"] == locale for item in self.data["variants"]):
|
|
77
|
-
raise ValueError("variant slug and locale collision")
|
|
78
100
|
if canonical_variant_id and not any(item["id"] == canonical_variant_id for item in self.data["variants"]):
|
|
79
101
|
raise ValueError("canonical variant does not exist")
|
|
102
|
+
canonical_url = variant_canonical_url(slug, locale, market)
|
|
103
|
+
if any(item.get("canonicalUrl") == canonical_url for item in self.data["variants"]):
|
|
104
|
+
raise ValueError("variant market, locale and slug route collision")
|
|
105
|
+
family = canonical_variant_id or variant_id
|
|
106
|
+
if any(item.get("canonicalVariantId") == family and item.get("locale") == locale for item in self.data["variants"]):
|
|
107
|
+
raise ValueError("variant locale collision in translation cluster")
|
|
80
108
|
item = {"id": variant_id, "serviceId": service_id, "variantType": variant_type, "locale": locale,
|
|
81
109
|
"market": market, "slug": slug, "title": title, "facts": facts,
|
|
82
110
|
"clusterLinks": sorted(set(cluster_links or [])), "layoutFamily": layout_family,
|
|
83
|
-
"canonicalVariantId":
|
|
84
|
-
"hreflang": {locale:
|
|
111
|
+
"canonicalVariantId": family, "canonicalUrl": canonical_url,
|
|
112
|
+
"hreflang": {locale: canonical_url}, "sourceRevision": source_revision,
|
|
85
113
|
"provenance": {"operation": "create", "actor": actor, "sourceRevision": source_revision},
|
|
86
114
|
"status": "draft", "createdAt": now(), "updatedAt": now()}
|
|
87
115
|
self.data["variants"].append(item); self._event("variant.created", variant_id, actor, "draft created"); self._save()
|
|
@@ -120,6 +148,7 @@ class ServiceVariantStore:
|
|
|
120
148
|
raise ValueError("canonical source must be published before a translated variant")
|
|
121
149
|
old = item["status"]; item["status"] = target; item["updatedAt"] = now()
|
|
122
150
|
item.setdefault("approval", []).append({"from": old, "to": target, "actor": actor, "reason": reason, "at": now()})
|
|
151
|
+
self._refresh_hreflang(item["canonicalVariantId"])
|
|
123
152
|
self._event("variant." + target, variant_id, actor, reason); self._save(); return item
|
|
124
153
|
|
|
125
154
|
def plan_slug_change(self, variant_id: str, new_slug: str, actor: str) -> dict:
|
|
@@ -128,17 +157,29 @@ class ServiceVariantStore:
|
|
|
128
157
|
raise ValueError("published URL changes require owner approval")
|
|
129
158
|
if not new_slug.startswith(item["variantType"] + "/") or not SLUG_WORD.fullmatch(new_slug.split("/", 1)[-1]):
|
|
130
159
|
raise ValueError("new slug violates variant grammar")
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
160
|
+
new_url = variant_canonical_url(new_slug, item["locale"], item["market"])
|
|
161
|
+
if any(v.get("canonicalUrl") == new_url for v in self.data["variants"] if v["id"] != variant_id):
|
|
162
|
+
raise ValueError("new market, locale and slug route collides with another variant")
|
|
163
|
+
redirect = {"from": item["canonicalUrl"], "to": new_url, "ownerApproval": actor, "status": "planned"}
|
|
134
164
|
self.data["redirects"].append(redirect); self._save(); return redirect
|
|
135
165
|
|
|
136
166
|
def validate(self, variant_id: str, source_text: str | None = None, render_report: dict | None = None) -> dict:
|
|
137
167
|
item = self._find(variant_id); errors = []
|
|
168
|
+
expected_url = variant_canonical_url(item["slug"], item["locale"], item["market"])
|
|
169
|
+
if item.get("canonicalUrl") != expected_url:
|
|
170
|
+
errors.append("canonical URL must include market, locale and variant slug")
|
|
138
171
|
if not item["clusterLinks"] and item["variantType"] != "location": errors.append("cluster links required")
|
|
139
172
|
if source_text is not None and similarity(source_text, item["title"] + " " + " ".join(str(f["value"]) for f in item["facts"])) < 0.1:
|
|
140
173
|
errors.append("variant content has no meaningful similarity to source")
|
|
141
174
|
if item["canonicalVariantId"] != item["id"] and item["canonicalVariantId"] not in {v["id"] for v in self.data["variants"]}: errors.append("canonical relation missing")
|
|
175
|
+
if item.get("status") == "published":
|
|
176
|
+
expected_hreflang = {
|
|
177
|
+
peer["locale"]: peer["canonicalUrl"]
|
|
178
|
+
for peer in self._cluster(item["canonicalVariantId"])
|
|
179
|
+
if peer.get("status") == "published"
|
|
180
|
+
}
|
|
181
|
+
if item.get("hreflang") != expected_hreflang:
|
|
182
|
+
errors.append("published hreflang links must be reciprocal published counterparts")
|
|
142
183
|
if render_report is not None:
|
|
143
184
|
if render_report.get("schemaVersion") != "maggie-service-variant-render.v1" or render_report.get("passed") is not True:
|
|
144
185
|
errors.append("render report is not passed")
|
package/package.json
CHANGED
|
@@ -25,8 +25,10 @@ when a title changes.
|
|
|
25
25
|
"title": "Signature Scalp Ritual",
|
|
26
26
|
"description": "Provider-supplied factual description.",
|
|
27
27
|
"category": {"level1": "Head Spa", "level2": "Scalp Treatments"},
|
|
28
|
-
|
|
28
|
+
"variants": [{
|
|
29
29
|
"id": "service-123:60",
|
|
30
|
+
"idSource": "provider",
|
|
31
|
+
"variantSource": "native",
|
|
30
32
|
"title": "60 minutes",
|
|
31
33
|
"durationMinutes": 60,
|
|
32
34
|
"price": {"amountMinor": 9500, "currency": "GBP", "display": "£95.00"}
|
|
@@ -147,3 +149,36 @@ Provider adapters may add `raw` evidence in a private local snapshot, but
|
|
|
147
149
|
public page generation consumes only the canonical fields above. A sync must
|
|
148
150
|
preserve removed records as `archived` with `removedAt` and must output a
|
|
149
151
|
change report.
|
|
152
|
+
|
|
153
|
+
## Provider variant capability matrix
|
|
154
|
+
|
|
155
|
+
Before service pages are published, each provider declares its variant
|
|
156
|
+
behavior in a project-owned JSON file. The declaration is validated against a
|
|
157
|
+
sanitized fixture and emits `maggie-provider-variant-capabilities.v1`:
|
|
158
|
+
|
|
159
|
+
```json
|
|
160
|
+
{
|
|
161
|
+
"schemaVersion": "maggie-provider-capabilities.v1",
|
|
162
|
+
"providers": [{
|
|
163
|
+
"provider": "fresha",
|
|
164
|
+
"variantMode": "native",
|
|
165
|
+
"priceSemantics": "minor-unit-and-display",
|
|
166
|
+
"stableIdConfidence": "high"
|
|
167
|
+
}]
|
|
168
|
+
}
|
|
169
|
+
```
|
|
170
|
+
|
|
171
|
+
`variantMode` is one of `native`, `derived-offers`, `single-fallback`, or
|
|
172
|
+
`blocked`. `priceSemantics` records whether structured minor units and display
|
|
173
|
+
prices are available. `stableIdConfidence` must be justified by provider or
|
|
174
|
+
derived ID evidence. The audit records service/variant counts, observed source
|
|
175
|
+
types, fixture name and hash, and validation errors without copying fixture
|
|
176
|
+
rows or provider credentials. Run it with:
|
|
177
|
+
|
|
178
|
+
```bash
|
|
179
|
+
maggie service capability-audit \
|
|
180
|
+
--project . \
|
|
181
|
+
--catalogue .maggie/booking/services.json \
|
|
182
|
+
--capabilities-file .maggie/booking/provider-capabilities.json \
|
|
183
|
+
--fixture .maggie/booking/fixtures/provider.json
|
|
184
|
+
```
|