@topy-ai/maggie 0.6.9 → 0.7.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +56 -4
- package/bin/maggie.js +11 -7
- package/bundled-references/blog-translation-ingestion.md +52 -0
- package/bundled-references/maggiedash-dashboard-ui.md +28 -0
- package/bundled-skills/maggie-blog/SKILL.md +28 -0
- package/bundled-skills/maggie-content-localization/SKILL.md +49 -0
- package/bundled-skills/maggie-dash/SKILL.md +52 -0
- package/bundled-skills/maggie-deployment/SKILL.md +17 -0
- package/bundled-skills/maggie-design/SKILL.md +11 -0
- package/bundled-skills/maggie-memory/SKILL.md +7 -0
- package/bundled-skills/maggie-ops/SKILL.md +27 -0
- package/bundled-skills/maggie-seo-geo/SKILL.md +95 -1
- package/bundled-skills/maggie-service-booking/SKILL.md +16 -0
- package/bundled-templates/maggiedash/README.md +4 -0
- package/bundled-templates/maggiedash/dashboard-ui-contract.json +31 -0
- package/bundled-tools/clis/maggie_analytics.py +14 -1
- package/bundled-tools/clis/maggie_blog.py +19 -1
- package/bundled-tools/clis/maggie_browser_audit.py +78 -0
- package/bundled-tools/clis/maggie_dash.py +101 -0
- package/bundled-tools/clis/maggie_deployment.py +43 -0
- package/bundled-tools/clis/maggie_feedback.py +16 -1
- package/bundled-tools/clis/maggie_localization.py +31 -0
- package/bundled-tools/clis/maggie_memory.py +4 -1
- package/bundled-tools/clis/maggie_ops.py +21 -1
- package/bundled-tools/clis/maggie_service_booking.py +6 -3
- package/bundled-tools/clis/maggie_sitemap.py +16 -2
- package/bundled-tools/clis/site_audit.py +93 -9
- package/bundled-tools/runtime/analytics_traffic.py +30 -0
- package/bundled-tools/runtime/browser_behavior.py +35 -0
- package/bundled-tools/runtime/browser_geometry.js +31 -0
- package/bundled-tools/runtime/content_localization.py +63 -1
- package/bundled-tools/runtime/dependency_lock.py +43 -0
- package/bundled-tools/runtime/integration_state.py +17 -0
- package/bundled-tools/runtime/localization_runner.py +113 -0
- package/bundled-tools/runtime/maggie_blog.py +66 -1
- package/bundled-tools/runtime/maggie_dash_store.py +150 -6
- package/bundled-tools/runtime/maggie_dash_ui.py +60 -0
- package/bundled-tools/runtime/maggie_memory.py +7 -2
- package/bundled-tools/runtime/maggie_sitemap.py +50 -4
- package/bundled-tools/runtime/route_imports.py +51 -0
- package/bundled-tools/runtime/seed_evidence.py +25 -0
- package/bundled-tools/runtime/service_variants.py +152 -0
- package/bundled-tools/runtime/site_baseline.py +60 -0
- package/package.json +1 -1
- package/references/blog-translation-ingestion.md +52 -0
- package/references/maggiedash-dashboard-ui.md +28 -0
|
@@ -10,7 +10,7 @@ import sys
|
|
|
10
10
|
from pathlib import Path
|
|
11
11
|
|
|
12
12
|
sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "runtime"))
|
|
13
|
-
from maggie_sitemap import build_plan, parse_routes, validate_plan_data # noqa: E402
|
|
13
|
+
from maggie_sitemap import agent_files, build_plan, parse_routes, validate_plan_data # noqa: E402
|
|
14
14
|
|
|
15
15
|
|
|
16
16
|
def write_json(path: str, value: dict) -> None:
|
|
@@ -31,6 +31,12 @@ def main() -> int:
|
|
|
31
31
|
plan.add_argument("--output", required=True)
|
|
32
32
|
validate = sub.add_parser("validate")
|
|
33
33
|
validate.add_argument("--plan", required=True)
|
|
34
|
+
validate.add_argument("--strict-semantic", action="store_true")
|
|
35
|
+
agent = sub.add_parser("agent-files")
|
|
36
|
+
agent.add_argument("--origin", required=True)
|
|
37
|
+
agent.add_argument("--routes-file", required=True)
|
|
38
|
+
agent.add_argument("--output-dir", required=True)
|
|
39
|
+
agent.add_argument("--locale", default="en")
|
|
34
40
|
apply = sub.add_parser("apply")
|
|
35
41
|
apply.add_argument("--plan", required=True)
|
|
36
42
|
apply.add_argument("--public-dir", required=True)
|
|
@@ -65,9 +71,17 @@ def main() -> int:
|
|
|
65
71
|
restored += 1
|
|
66
72
|
print(json.dumps({"status": "rolled-back", "restored": restored}, indent=2))
|
|
67
73
|
return 0
|
|
74
|
+
if args.command == "agent-files":
|
|
75
|
+
output = Path(args.output_dir)
|
|
76
|
+
output.mkdir(parents=True, exist_ok=True)
|
|
77
|
+
files = agent_files(parse_routes(Path(args.routes_file).read_text(encoding="utf-8")), args.origin, args.locale)
|
|
78
|
+
for name, content in files.items():
|
|
79
|
+
(output / name).write_text(content, encoding="utf-8")
|
|
80
|
+
print(json.dumps({"status": "generated", "locale": args.locale, "files": [str(output / name) for name in files]}, indent=2))
|
|
81
|
+
return 0
|
|
68
82
|
plan_value = json.loads(Path(args.plan).read_text(encoding="utf-8"))
|
|
69
83
|
if args.command == "validate":
|
|
70
|
-
result = validate_plan_data(plan_value)
|
|
84
|
+
result = validate_plan_data(plan_value, args.strict_semantic)
|
|
71
85
|
print(json.dumps(result, indent=2))
|
|
72
86
|
return 0 if result["status"] == "pass" else 1
|
|
73
87
|
if not args.confirm:
|
|
@@ -4,6 +4,7 @@
|
|
|
4
4
|
from __future__ import annotations
|
|
5
5
|
|
|
6
6
|
import argparse
|
|
7
|
+
import hashlib
|
|
7
8
|
import json
|
|
8
9
|
import re
|
|
9
10
|
import sys
|
|
@@ -15,6 +16,7 @@ from urllib.request import Request, urlopen
|
|
|
15
16
|
|
|
16
17
|
sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "runtime"))
|
|
17
18
|
from content_localization import MARKETS, parse_locale, valid_locale # noqa: E402
|
|
19
|
+
import site_baseline
|
|
18
20
|
|
|
19
21
|
|
|
20
22
|
class PageParser(HTMLParser):
|
|
@@ -33,9 +35,20 @@ class PageParser(HTMLParser):
|
|
|
33
35
|
self.images = []
|
|
34
36
|
self.hreflang = []
|
|
35
37
|
self.robots_directives = []
|
|
38
|
+
self.structure = []
|
|
39
|
+
self.visible_text = []
|
|
40
|
+
self.hidden_depth = 0
|
|
41
|
+
self.templates = set()
|
|
36
42
|
|
|
37
43
|
def handle_starttag(self, tag, attrs):
|
|
38
44
|
data = dict(attrs)
|
|
45
|
+
if data.get("data-template"):
|
|
46
|
+
self.templates.add(data["data-template"])
|
|
47
|
+
if tag in {"script", "style", "noscript"}:
|
|
48
|
+
self.hidden_depth += 1
|
|
49
|
+
if not self.hidden_depth:
|
|
50
|
+
self.structure.append(["start", tag, {key: data[key] for key in
|
|
51
|
+
("class", "id", "role", "href", "src", "data-template", "data-section", "data-i18n") if key in data}])
|
|
39
52
|
if tag == "html":
|
|
40
53
|
self.lang = data.get("lang", "")
|
|
41
54
|
if tag == "meta" and data.get("name"):
|
|
@@ -61,6 +74,10 @@ class PageParser(HTMLParser):
|
|
|
61
74
|
self.images.append({"src": data.get("src", ""), "alt": data.get("alt")})
|
|
62
75
|
|
|
63
76
|
def handle_endtag(self, tag):
|
|
77
|
+
if tag in {"script", "style", "noscript"}:
|
|
78
|
+
self.hidden_depth = max(0, self.hidden_depth - 1)
|
|
79
|
+
elif not self.hidden_depth:
|
|
80
|
+
self.structure.append(["end", tag])
|
|
64
81
|
if tag == "title":
|
|
65
82
|
self.in_title = False
|
|
66
83
|
if tag == "script" and self._jsonld is not None:
|
|
@@ -72,25 +89,35 @@ class PageParser(HTMLParser):
|
|
|
72
89
|
self._jsonld = None
|
|
73
90
|
|
|
74
91
|
def handle_data(self, data):
|
|
92
|
+
if not self.hidden_depth and data.strip():
|
|
93
|
+
self.visible_text.append(" ".join(data.split()))
|
|
75
94
|
if self.in_title:
|
|
76
95
|
self.title += data.strip()
|
|
77
96
|
if self._jsonld is not None:
|
|
78
97
|
self._jsonld.append(data)
|
|
79
98
|
|
|
80
99
|
|
|
81
|
-
def fetch(url: str) -> tuple[int, str, str]:
|
|
100
|
+
def fetch(url: str, evidence: dict | None = None) -> tuple[int, str, str]:
|
|
82
101
|
request = Request(url, headers={"User-Agent": "AI-CMO-Skills-Audit/0.1"})
|
|
83
102
|
with urlopen(request, timeout=15) as response:
|
|
84
|
-
|
|
103
|
+
raw = response.read(2_000_001)
|
|
104
|
+
if len(raw) > 2_000_000:
|
|
105
|
+
raise ValueError("response exceeds audit size limit")
|
|
106
|
+
body = raw.decode(response.headers.get_content_charset() or "utf-8", "replace")
|
|
107
|
+
if evidence is not None:
|
|
108
|
+
evidence.update({"finalUrl": response.geturl(),
|
|
109
|
+
"xRobotsTag": response.headers.get_all("X-Robots-Tag", [])})
|
|
85
110
|
return response.status, response.headers.get_content_type(), body
|
|
86
111
|
|
|
87
112
|
|
|
88
|
-
def audit_page(url: str, html: str, status: int, content_type: str, expected_languages: set[str] | None = None) -> dict:
|
|
113
|
+
def audit_page(url: str, html: str, status: int, content_type: str, expected_languages: set[str] | None = None, response_evidence: dict | None = None) -> dict:
|
|
89
114
|
page = PageParser()
|
|
90
115
|
page.feed(html)
|
|
91
116
|
canonical = urljoin(url, page.canonical) if page.canonical else ""
|
|
92
|
-
|
|
117
|
+
response_evidence = response_evidence or {"finalUrl": url, "xRobotsTag": []}
|
|
118
|
+
robots = page.robots_directives + [value.lower() for value in response_evidence["xRobotsTag"]]
|
|
93
119
|
robots_tokens = [token.strip() for directive in robots for token in directive.split(",")]
|
|
120
|
+
robots_tokens = [token.rsplit(":", 1)[-1].strip() for token in robots_tokens]
|
|
94
121
|
return {
|
|
95
122
|
"url": url,
|
|
96
123
|
"status": status,
|
|
@@ -109,13 +136,45 @@ def audit_page(url: str, html: str, status: int, content_type: str, expected_lan
|
|
|
109
136
|
"canonical_value": canonical,
|
|
110
137
|
"image_count": len(page.images),
|
|
111
138
|
"language": parse_locale(page.lang)[0],
|
|
139
|
+
"declaredLocale": page.lang,
|
|
140
|
+
"template": next(iter(page.templates)) if len(page.templates) == 1 else "unknown",
|
|
141
|
+
"evidence": {"structural": "captured", "rendered": "not_run", "behavioral": "not_run"},
|
|
112
142
|
"hreflang": page.hreflang,
|
|
143
|
+
"contract": {
|
|
144
|
+
"response": response_evidence,
|
|
145
|
+
"title": page.title, "meta": page.meta, "canonical": canonical,
|
|
146
|
+
"lang": page.lang, "hreflang": page.hreflang, "robots": robots,
|
|
147
|
+
"jsonld": page.jsonld_values, "images": page.images,
|
|
148
|
+
"structureHash": hashlib.sha256(json.dumps(page.structure, sort_keys=True).encode()).hexdigest(),
|
|
149
|
+
"textHash": hashlib.sha256(" ".join(page.visible_text).encode()).hexdigest(),
|
|
150
|
+
},
|
|
113
151
|
"robots": not robots or not any(token in {"noindex", "none", "nofollow"} for token in robots_tokens),
|
|
114
152
|
"robots_directives": robots,
|
|
115
153
|
"robots_conflict": len({token for token in robots_tokens if token in {"index", "noindex", "follow", "nofollow", "none"}} & {"index", "noindex"}) > 1 or len({token for token in robots_tokens if token in {"follow", "nofollow", "none"}} & {"follow", "nofollow"}) > 1,
|
|
116
154
|
}
|
|
117
155
|
|
|
118
156
|
|
|
157
|
+
def summarize_crawl(pages: list[dict]) -> dict:
|
|
158
|
+
"""Count captured evidence, preserving regional/script locale distinctions."""
|
|
159
|
+
groups = {}
|
|
160
|
+
for page in pages:
|
|
161
|
+
locale = page.get("declaredLocale") or "unknown"
|
|
162
|
+
template = page.get("template") or "unknown"
|
|
163
|
+
key = (locale, template)
|
|
164
|
+
group = groups.setdefault(key, {"locale": locale, "template": template,
|
|
165
|
+
"total": 0, "passed": 0, "failed": 0, "failedUrls": [],
|
|
166
|
+
"evidence": {"structural": "captured", "rendered": "not_run", "behavioral": "not_run"}})
|
|
167
|
+
group["total"] += 1
|
|
168
|
+
if page.get("passed") is True:
|
|
169
|
+
group["passed"] += 1
|
|
170
|
+
else:
|
|
171
|
+
group["failed"] += 1
|
|
172
|
+
group["failedUrls"].append(page["url"])
|
|
173
|
+
for group in groups.values():
|
|
174
|
+
group["failedUrls"].sort()
|
|
175
|
+
return {"byLocaleTemplate": [groups[key] for key in sorted(groups)]}
|
|
176
|
+
|
|
177
|
+
|
|
119
178
|
def language_set(value: str | None) -> set[str]:
|
|
120
179
|
if not value:
|
|
121
180
|
return set()
|
|
@@ -178,7 +237,17 @@ def main() -> int:
|
|
|
178
237
|
parser.add_argument("--markets", help="comma-separated markets: global,uk,us")
|
|
179
238
|
parser.add_argument("--check-hreflang", action="store_true")
|
|
180
239
|
parser.add_argument("--check-translation-completeness", action="store_true")
|
|
240
|
+
baseline_args = parser.add_mutually_exclusive_group()
|
|
241
|
+
baseline_args.add_argument("--save-baseline", type=Path, help="create a new reviewed contract from a passing complete crawl")
|
|
242
|
+
baseline_args.add_argument("--baseline", type=Path, help="fail on differences from a reviewed contract")
|
|
243
|
+
parser.add_argument("--reviewer", help="required for --save-baseline")
|
|
181
244
|
args = parser.parse_args()
|
|
245
|
+
if args.max_pages < 1:
|
|
246
|
+
parser.error("max-pages must be positive")
|
|
247
|
+
if (args.save_baseline or args.baseline) and not args.crawl:
|
|
248
|
+
parser.error("baseline operations require --crawl")
|
|
249
|
+
if args.save_baseline and not (args.reviewer or "").strip():
|
|
250
|
+
parser.error("save-baseline requires --reviewer")
|
|
182
251
|
try:
|
|
183
252
|
expected_languages = language_set(args.languages)
|
|
184
253
|
markets = {item.strip() for item in (args.markets or "").split(",") if item.strip()}
|
|
@@ -232,11 +301,14 @@ def main() -> int:
|
|
|
232
301
|
if args.crawl:
|
|
233
302
|
try:
|
|
234
303
|
urls, sitemap_violations = sitemap_urls(base)
|
|
304
|
+
crawl["discovered_url_count"] = len(urls)
|
|
305
|
+
crawl["complete"] = len(urls) <= args.max_pages and not sitemap_violations
|
|
235
306
|
urls = urls[: args.max_pages]
|
|
236
307
|
for page_url in urls:
|
|
237
308
|
try:
|
|
238
|
-
|
|
239
|
-
|
|
309
|
+
response_evidence = {}
|
|
310
|
+
status, content_type, body = fetch(page_url, response_evidence)
|
|
311
|
+
page_checks = audit_page(page_url, body, status, content_type, expected_languages, response_evidence)
|
|
240
312
|
if args.check_hreflang or args.check_translation_completeness:
|
|
241
313
|
parsed = PageParser(); parsed.feed(body)
|
|
242
314
|
page_checks["hreflang_check"] = hreflang_check(page_url, parsed, expected_languages) if args.check_hreflang else {"ok": True, "links": parsed.hreflang}
|
|
@@ -244,18 +316,30 @@ def main() -> int:
|
|
|
244
316
|
page_checks["translation_completeness"] = {"ok": all(lang in {item["lang"] for item in parsed.hreflang} for lang in expected_languages), "expected": sorted(expected_languages)}
|
|
245
317
|
page_checks["passed"] = all(
|
|
246
318
|
page_checks[key]
|
|
247
|
-
for key in ("ok", "title", "h1", "canonical", "description", "locale", "open_graph", "twitter_card", "jsonld", "entity_jsonld", "images", "robots"
|
|
248
|
-
) and (not args.check_hreflang or page_checks["hreflang_check"]["ok"]) and (not args.check_translation_completeness or page_checks.get("translation_completeness", {}).get("ok", False))
|
|
319
|
+
for key in ("ok", "title", "h1", "canonical", "description", "locale", "open_graph", "twitter_card", "jsonld", "entity_jsonld", "images", "robots")
|
|
320
|
+
) and not page_checks["robots_conflict"] and (not args.check_hreflang or page_checks["hreflang_check"]["ok"]) and (not args.check_translation_completeness or page_checks.get("translation_completeness", {}).get("ok", False))
|
|
249
321
|
except Exception as exc:
|
|
250
322
|
page_checks = {"url": page_url, "passed": False, "error": type(exc).__name__}
|
|
251
323
|
crawl["pages"].append(page_checks)
|
|
252
324
|
crawl["sitemap_loc_violations"] = sitemap_violations
|
|
253
|
-
crawl["passed"] =
|
|
325
|
+
crawl["passed"] = crawl["complete"] and bool(urls) and len(crawl["pages"]) == len(urls) and all(item["passed"] for item in crawl["pages"])
|
|
254
326
|
crawl["url_count"] = len(urls)
|
|
255
327
|
except Exception as exc:
|
|
256
328
|
crawl = {"enabled": True, "passed": False, "error": type(exc).__name__, "pages": []}
|
|
257
329
|
|
|
258
330
|
result = {"url": base, "environment": args.environment, "checks": checks, "crawl": crawl, "passed": all(v.get("ok", False) for v in checks.values()) and crawl["passed"]}
|
|
331
|
+
result["evidence"] = {"structural": "captured", "rendered": "not_run", "behavioral": "not_run"}
|
|
332
|
+
crawl["summary"] = summarize_crawl(crawl["pages"])
|
|
333
|
+
try:
|
|
334
|
+
if args.baseline:
|
|
335
|
+
result["baseline"] = site_baseline.compare(json.loads(args.baseline.read_text()), result)
|
|
336
|
+
result["passed"] = result["passed"] and result["baseline"]["passed"]
|
|
337
|
+
if args.save_baseline:
|
|
338
|
+
site_baseline.save(args.save_baseline, site_baseline.snapshot(result, args.reviewer))
|
|
339
|
+
result["baseline"] = {"passed": True, "saved": str(args.save_baseline)}
|
|
340
|
+
except (OSError, ValueError, TypeError) as exc:
|
|
341
|
+
result["baseline"] = {"passed": False, "error": str(exc)}
|
|
342
|
+
result["passed"] = False
|
|
259
343
|
if args.output:
|
|
260
344
|
output = Path(args.output).resolve()
|
|
261
345
|
output.parent.mkdir(parents=True, exist_ok=True)
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
"""Safe classification of known first-party/toolchain traffic."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
from collections import Counter
|
|
7
|
+
from typing import Iterable
|
|
8
|
+
|
|
9
|
+
TOOL_USER_AGENT = re.compile(r"(?:maggie|playwright|puppeteer|selenium|lighthouse|headlesschrome|synthetic|test[-_ ]?sweep)", re.I)
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def is_tool_traffic(event: dict) -> bool:
|
|
13
|
+
if str(event.get("maggieTest") or "").lower() == "true":
|
|
14
|
+
return True
|
|
15
|
+
if str(event.get("xMaggieTest") or "").lower() == "true":
|
|
16
|
+
return True
|
|
17
|
+
return bool(TOOL_USER_AGENT.search(str(event.get("userAgent") or "")))
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def audit_events(events: Iterable[dict]) -> dict:
|
|
21
|
+
totals = Counter()
|
|
22
|
+
excluded = Counter()
|
|
23
|
+
included = 0
|
|
24
|
+
for event in events:
|
|
25
|
+
totals[str(event.get("event") or "unknown")] += 1
|
|
26
|
+
if is_tool_traffic(event):
|
|
27
|
+
excluded[str(event.get("event") or "unknown")] += 1
|
|
28
|
+
else:
|
|
29
|
+
included += 1
|
|
30
|
+
return {"schemaVersion": "maggie-analytics-traffic.v1", "total": sum(totals.values()), "included": included, "excluded": sum(excluded.values()), "byEvent": dict(sorted(totals.items())), "excludedByEvent": dict(sorted(excluded.items())), "policy": "exclude only explicit Maggie/toolchain markers; no IP or identity inference"}
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
"""Compare actual browser samples against explicit route expectations."""
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
def validate_samples(samples: list[dict], selectors: list[str], sticky: list[str]) -> dict:
|
|
5
|
+
errors = []
|
|
6
|
+
if len(samples) < 2:
|
|
7
|
+
return {"passed": False, "errors": ["at least two scroll samples required"]}
|
|
8
|
+
first = samples[0]
|
|
9
|
+
for sample in samples:
|
|
10
|
+
if sample.get("viewport") != first.get("viewport") or sample.get("url") != first.get("url"):
|
|
11
|
+
errors.append("samples must share URL and viewport")
|
|
12
|
+
if sample.get("horizontalOverflow"):
|
|
13
|
+
errors.append("horizontal overflow")
|
|
14
|
+
elements = {element["selector"]: element for element in sample.get("elements", [])}
|
|
15
|
+
for selector in selectors:
|
|
16
|
+
element = elements.get(selector, {})
|
|
17
|
+
if not element.get("found") or not element.get("visible"):
|
|
18
|
+
errors.append(f"required element missing or hidden: {selector}")
|
|
19
|
+
if samples[-1].get("scroll", {}).get("y", 0) <= first.get("scroll", {}).get("y", 0):
|
|
20
|
+
errors.append("scroll did not advance; behavior unverified")
|
|
21
|
+
for selector in sticky:
|
|
22
|
+
for sample in samples[1:]:
|
|
23
|
+
element = next((item for item in sample.get("elements", []) if item.get("selector") == selector), {})
|
|
24
|
+
if element.get("position") not in {"sticky", "fixed"}:
|
|
25
|
+
errors.append(f"expected sticky/fixed positioning: {selector}")
|
|
26
|
+
continue
|
|
27
|
+
top = element.get("rect", {}).get("top")
|
|
28
|
+
try:
|
|
29
|
+
inset = float(element.get("insetTop", "auto").removesuffix("px"))
|
|
30
|
+
except (ValueError, AttributeError):
|
|
31
|
+
errors.append(f"sticky top inset unmeasurable: {selector}")
|
|
32
|
+
continue
|
|
33
|
+
if not isinstance(top, (int, float)) or abs(top - inset) > 2:
|
|
34
|
+
errors.append(f"sticky element left expected top inset: {selector}")
|
|
35
|
+
return {"passed": not errors, "errors": sorted(set(errors))}
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
// Read-only browser measurements. The caller supplies selectors and scrolls
|
|
2
|
+
// between samples; never infer scrolling behavior from position:sticky alone.
|
|
3
|
+
((selectors = window.__maggieAuditSelectors || []) => {
|
|
4
|
+
const visible = element => {
|
|
5
|
+
const rect = element.getBoundingClientRect();
|
|
6
|
+
if (!rect.width || !rect.height) return false;
|
|
7
|
+
for (let node = element; node; node = node.parentElement) {
|
|
8
|
+
const style = getComputedStyle(node);
|
|
9
|
+
if (style.display === 'none' || style.visibility === 'hidden' || Number(style.opacity) === 0) return false;
|
|
10
|
+
}
|
|
11
|
+
return true;
|
|
12
|
+
};
|
|
13
|
+
return {
|
|
14
|
+
url: location.href,
|
|
15
|
+
viewport: {width: innerWidth, height: innerHeight},
|
|
16
|
+
scroll: {x: scrollX, y: scrollY, maxY: Math.max(0, document.documentElement.scrollHeight - innerHeight)},
|
|
17
|
+
horizontalOverflow: document.documentElement.scrollWidth > innerWidth + 1,
|
|
18
|
+
elements: selectors.map(selector => {
|
|
19
|
+
const element = document.querySelector(selector);
|
|
20
|
+
if (!element) return {selector, found: false, visible: false};
|
|
21
|
+
const rect = element.getBoundingClientRect();
|
|
22
|
+
const style = getComputedStyle(element);
|
|
23
|
+
const parent = element.parentElement?.getBoundingClientRect();
|
|
24
|
+
return {selector, found: true, visible: visible(element),
|
|
25
|
+
rect: {top: rect.top, bottom: rect.bottom, left: rect.left, right: rect.right, width: rect.width, height: rect.height},
|
|
26
|
+
position: style.position, insetTop: style.top,
|
|
27
|
+
parentRect: parent ? {top: parent.top, bottom: parent.bottom, height: parent.height} : null,
|
|
28
|
+
hiddenDescendants: [...element.querySelectorAll('a,button,input,select')].filter(child => !visible(child)).length};
|
|
29
|
+
})
|
|
30
|
+
};
|
|
31
|
+
})()
|
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
from __future__ import annotations
|
|
4
4
|
|
|
5
5
|
import re
|
|
6
|
-
from typing import Any
|
|
6
|
+
from typing import Any, Iterable
|
|
7
7
|
|
|
8
8
|
# Deliberately broad, dependency-free registry of the major web/content
|
|
9
9
|
# languages. Script variants remain distinct where they affect copy, search,
|
|
@@ -19,6 +19,68 @@ OPERATIONS = {"translate", "polish", "rewrite", "localise", "rebrand", "manual_e
|
|
|
19
19
|
PROTECTED_FIELDS = {"price", "currency", "rating", "provider", "providerId", "bookingUrl", "paymentUrl", "legalClaims", "healthClaims", "contentId", "slug", "canonicalUrl", "translationGroupId"}
|
|
20
20
|
|
|
21
21
|
|
|
22
|
+
class TranslationIndex:
|
|
23
|
+
"""One canonical index for translated records and their public wording.
|
|
24
|
+
|
|
25
|
+
Hosts may persist this structure wherever they keep content, but they must
|
|
26
|
+
build it once and reject conflicting duplicate `(contentId, locale)` rows.
|
|
27
|
+
This prevents a registry and a second dictionary from silently disagreeing.
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
def __init__(self, records: Iterable[dict[str, Any]] = ()):
|
|
31
|
+
self._records: dict[tuple[str, str], dict[str, Any]] = {}
|
|
32
|
+
self.conflicts: list[dict[str, Any]] = []
|
|
33
|
+
for record in records:
|
|
34
|
+
self.add(record)
|
|
35
|
+
|
|
36
|
+
def add(self, record: dict[str, Any]) -> None:
|
|
37
|
+
content_id = str(record.get("contentId") or "").strip()
|
|
38
|
+
locale = str(record.get("locale") or "").strip()
|
|
39
|
+
if not content_id or not locale:
|
|
40
|
+
raise ValueError("translation index records require contentId and locale")
|
|
41
|
+
key = (content_id, locale)
|
|
42
|
+
previous = self._records.get(key)
|
|
43
|
+
if previous and previous != record:
|
|
44
|
+
self.conflicts.append({"key": key, "existing": previous, "incoming": record})
|
|
45
|
+
raise ValueError(f"conflicting translation record for {content_id}/{locale}")
|
|
46
|
+
self._records[key] = dict(record)
|
|
47
|
+
|
|
48
|
+
def get(self, content_id: str, locale: str) -> dict[str, Any] | None:
|
|
49
|
+
record = self._records.get((content_id, locale))
|
|
50
|
+
return dict(record) if record else None
|
|
51
|
+
|
|
52
|
+
def records(self) -> list[dict[str, Any]]:
|
|
53
|
+
return [dict(self._records[key]) for key in sorted(self._records)]
|
|
54
|
+
|
|
55
|
+
def as_dict(self) -> dict[str, dict[str, Any]]:
|
|
56
|
+
return {f"{content_id}:{locale}": dict(record) for (content_id, locale), record in sorted(self._records.items())}
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def is_translated_path(path: str, exact_paths: Iterable[str] = (), prefixes: Iterable[str] = ()) -> bool:
|
|
60
|
+
"""Use exact routes and normalized prefix routes with identical semantics."""
|
|
61
|
+
normalized = "/" + str(path or "").lstrip("/")
|
|
62
|
+
normalized = normalized.rstrip("/") or "/"
|
|
63
|
+
exact = {("/" + str(item).lstrip("/")).rstrip("/") or "/" for item in exact_paths}
|
|
64
|
+
normalized_prefixes = {("/" + str(item).lstrip("/")).rstrip("/") or "/" for item in prefixes}
|
|
65
|
+
return normalized in exact or any(normalized == prefix or normalized.startswith(prefix + "/") for prefix in normalized_prefixes)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def should_localize_path(path: str, locale_prefixes: Iterable[str], static_extensions: Iterable[str] = (".txt", ".md", ".xml", ".json")) -> bool:
|
|
69
|
+
"""Locale middleware only handles localized routes, never root static files."""
|
|
70
|
+
normalized = "/" + str(path or "").lstrip("/")
|
|
71
|
+
if any(normalized.lower().endswith(extension.lower()) for extension in static_extensions):
|
|
72
|
+
return False
|
|
73
|
+
return is_translated_path(normalized, prefixes=locale_prefixes)
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def rewrite_once(request_key: str, seen: set[str]) -> bool:
|
|
77
|
+
"""Return true only for the first rewrite observation of a request."""
|
|
78
|
+
if request_key in seen:
|
|
79
|
+
return False
|
|
80
|
+
seen.add(request_key)
|
|
81
|
+
return True
|
|
82
|
+
|
|
83
|
+
|
|
22
84
|
def parse_locale(value: Any) -> tuple[str | None, str | None, str | None]:
|
|
23
85
|
if not isinstance(value, str):
|
|
24
86
|
return None, None, None
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
"""Detect package-manager drift before an npm-ci deployment."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
import json
|
|
4
|
+
import re
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def package_dependencies(project: Path) -> set[str]:
|
|
9
|
+
data = json.loads((project / "package.json").read_text(encoding="utf-8"))
|
|
10
|
+
return set(data.get("dependencies", {})) | set(data.get("devDependencies", {})) | set(data.get("optionalDependencies", {}))
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def npm_lock_dependencies(path: Path) -> set[str]:
|
|
14
|
+
data = json.loads(path.read_text(encoding="utf-8"))
|
|
15
|
+
packages = data.get("packages", {})
|
|
16
|
+
names = {key.removeprefix("node_modules/") for key in packages
|
|
17
|
+
if key.startswith("node_modules/") and "/node_modules/" not in key}
|
|
18
|
+
if not packages and isinstance(data.get("dependencies"), dict):
|
|
19
|
+
names = set(data["dependencies"])
|
|
20
|
+
return names
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def pnpm_lock_dependencies(path: Path) -> set[str]:
|
|
24
|
+
text = path.read_text(encoding="utf-8", errors="replace")
|
|
25
|
+
return {match.group(1) for match in re.finditer(r"^\s{4,}/?(@[^/\s]+/[^/\s]+|[A-Za-z0-9_.-]+)@[^:]+:", text, re.M)}
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def audit(project: Path) -> dict:
|
|
29
|
+
errors = []
|
|
30
|
+
try:
|
|
31
|
+
declared = package_dependencies(project)
|
|
32
|
+
except (OSError, ValueError, json.JSONDecodeError) as error:
|
|
33
|
+
return {"passed": False, "errors": [f"package.json: {error}"]}
|
|
34
|
+
npm = project / "package-lock.json"
|
|
35
|
+
pnpm = project / "pnpm-lock.yaml"
|
|
36
|
+
npm_names = npm_lock_dependencies(npm) if npm.exists() else set()
|
|
37
|
+
pnpm_names = pnpm_lock_dependencies(pnpm) if pnpm.exists() else set()
|
|
38
|
+
if not npm.exists(): errors.append("package-lock.json is required by npm ci")
|
|
39
|
+
missing = sorted(declared - npm_names) if npm.exists() else sorted(declared)
|
|
40
|
+
if missing: errors.append("package-lock missing declared dependencies: " + ", ".join(missing))
|
|
41
|
+
return {"schemaVersion": "maggie-dependency-lock-audit.v1", "passed": not errors,
|
|
42
|
+
"errors": errors, "declared": sorted(declared), "npmLock": sorted(npm_names),
|
|
43
|
+
"pnpmLock": sorted(pnpm_names), "lockfiles": {"npm": npm.exists(), "pnpm": pnpm.exists()}}
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
"""Explicit state model for optional Search/AI visibility integrations."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def integration_state(*, configured: bool, consent_required: bool = False, consent: bool = False, authorized: bool = False, error: str | None = None) -> dict:
|
|
7
|
+
if error:
|
|
8
|
+
status = "error"
|
|
9
|
+
elif not configured:
|
|
10
|
+
status = "not-configured"
|
|
11
|
+
elif consent_required and not consent:
|
|
12
|
+
status = "awaiting-consent"
|
|
13
|
+
elif not authorized:
|
|
14
|
+
status = "awaiting-authorization"
|
|
15
|
+
else:
|
|
16
|
+
status = "ready"
|
|
17
|
+
return {"status": status, "configured": configured, "consentRequired": consent_required, "consent": consent, "authorized": authorized, "reason": error or status}
|
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
"""Resumable translation batches; provider calls are supplied by the adapter."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import hashlib
|
|
5
|
+
import json
|
|
6
|
+
import os
|
|
7
|
+
import subprocess
|
|
8
|
+
import tempfile
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def process_adapter(command: list[str], project: Path, timeout: int = 120):
|
|
13
|
+
if not isinstance(command, list) or not command or any(not isinstance(part, str) or not part for part in command):
|
|
14
|
+
raise ValueError("adapter-command must be a nonempty JSON argv array")
|
|
15
|
+
if timeout < 1:
|
|
16
|
+
raise ValueError("timeout must be positive")
|
|
17
|
+
def adapter(batch, contract):
|
|
18
|
+
try:
|
|
19
|
+
result = subprocess.run(command, input=json.dumps({
|
|
20
|
+
"schemaVersion": "maggie-provider-request.v1", "strings": batch,
|
|
21
|
+
"contract": contract}, ensure_ascii=False), text=True, capture_output=True,
|
|
22
|
+
timeout=timeout, check=False, cwd=project.resolve())
|
|
23
|
+
except (OSError, subprocess.TimeoutExpired):
|
|
24
|
+
raise OSError("provider adapter could not execute or timed out; retry to resume") from None
|
|
25
|
+
if result.returncode:
|
|
26
|
+
raise OSError("provider adapter failed; retry to resume (provider logs suppressed)")
|
|
27
|
+
return result.stdout
|
|
28
|
+
return adapter
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def checkpoint(path: Path, state: dict) -> None:
|
|
32
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
33
|
+
descriptor, name = tempfile.mkstemp(prefix=".translation-", dir=path.parent)
|
|
34
|
+
try:
|
|
35
|
+
with os.fdopen(descriptor, "w", encoding="utf-8") as stream:
|
|
36
|
+
json.dump(state, stream, ensure_ascii=False, indent=2)
|
|
37
|
+
stream.flush()
|
|
38
|
+
os.fsync(stream.fileno())
|
|
39
|
+
os.replace(name, path)
|
|
40
|
+
finally:
|
|
41
|
+
if os.path.exists(name):
|
|
42
|
+
os.unlink(name)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def generate(strings: list[dict], contract: dict, output: Path, adapter,
|
|
46
|
+
character_budget: int = 8000) -> dict:
|
|
47
|
+
"""Adapter accepts (batch, contract) and returns an exact ID-to-text map.
|
|
48
|
+
|
|
49
|
+
Invalid/truncated output splits a batch. Provider exceptions propagate;
|
|
50
|
+
successfully checkpointed batches are reusable on the next invocation.
|
|
51
|
+
"""
|
|
52
|
+
if type(character_budget) is not int or character_budget < 1:
|
|
53
|
+
raise ValueError("character_budget must be positive")
|
|
54
|
+
if not strings or any(not isinstance(item, dict) or not isinstance(item.get("id"), str)
|
|
55
|
+
or not item["id"] or not isinstance(item.get("text"), str)
|
|
56
|
+
or not item["text"].strip() for item in strings):
|
|
57
|
+
raise ValueError("source strings require nonempty IDs and text")
|
|
58
|
+
if len({item["id"] for item in strings}) != len(strings):
|
|
59
|
+
raise ValueError("duplicate source IDs")
|
|
60
|
+
identity = hashlib.sha256(json.dumps({"strings": strings, "contract": contract},
|
|
61
|
+
sort_keys=True, ensure_ascii=False).encode()).hexdigest()
|
|
62
|
+
state = {"schemaVersion": "maggie-generation.v1", "identity": identity,
|
|
63
|
+
"translations": {}, "status": "pending", "publish": False}
|
|
64
|
+
if output.exists():
|
|
65
|
+
state = json.loads(output.read_text(encoding="utf-8"))
|
|
66
|
+
if not isinstance(state, dict) or state.get("schemaVersion") != "maggie-generation.v1" or state.get("publish") is not False:
|
|
67
|
+
raise ValueError("invalid generation checkpoint")
|
|
68
|
+
if state.get("identity") != identity:
|
|
69
|
+
raise ValueError("source or generation contract changed; choose a new output")
|
|
70
|
+
translations = state.get("translations")
|
|
71
|
+
if not isinstance(translations, dict) or any(key not in {s["id"] for s in strings}
|
|
72
|
+
or not isinstance(value, str) or not value.strip() for key, value in translations.items()):
|
|
73
|
+
raise ValueError("invalid generation checkpoint")
|
|
74
|
+
|
|
75
|
+
def run(batch):
|
|
76
|
+
try:
|
|
77
|
+
result = adapter(batch, contract)
|
|
78
|
+
if isinstance(result, str):
|
|
79
|
+
result = json.loads(result)
|
|
80
|
+
if not isinstance(result, dict) or set(result) != {item["id"] for item in batch}:
|
|
81
|
+
raise ValueError("provider returned missing or unexpected IDs")
|
|
82
|
+
if any(not isinstance(value, str) or not value.strip() for value in result.values()):
|
|
83
|
+
raise ValueError("provider returned empty translations")
|
|
84
|
+
except (ValueError, json.JSONDecodeError):
|
|
85
|
+
if len(batch) == 1:
|
|
86
|
+
raise ValueError("provider output invalid for single source string") from None
|
|
87
|
+
midpoint = len(batch) // 2
|
|
88
|
+
run(batch[:midpoint])
|
|
89
|
+
run(batch[midpoint:])
|
|
90
|
+
return
|
|
91
|
+
state["translations"].update(result)
|
|
92
|
+
checkpoint(output, state)
|
|
93
|
+
|
|
94
|
+
pending = [item for item in strings if item["id"] not in state["translations"]]
|
|
95
|
+
state["status"] = "running"
|
|
96
|
+
checkpoint(output, state)
|
|
97
|
+
try:
|
|
98
|
+
batch, size = [], 0
|
|
99
|
+
for item in pending:
|
|
100
|
+
if batch and size + len(item["text"]) > character_budget:
|
|
101
|
+
run(batch)
|
|
102
|
+
batch, size = [], 0
|
|
103
|
+
batch.append(item)
|
|
104
|
+
size += len(item["text"])
|
|
105
|
+
if batch:
|
|
106
|
+
run(batch)
|
|
107
|
+
except Exception:
|
|
108
|
+
state["status"] = "failed"
|
|
109
|
+
checkpoint(output, state)
|
|
110
|
+
raise
|
|
111
|
+
state["status"] = "generated"
|
|
112
|
+
checkpoint(output, state)
|
|
113
|
+
return state
|