@topy-ai/maggie 0.6.8 → 0.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +39 -2
- package/README.zh-TW.md +1 -1
- package/bin/maggie.js +6 -2
- package/bundled-references/blog-translation-ingestion.md +52 -0
- package/bundled-skills/maggie-blog/SKILL.md +16 -0
- package/bundled-skills/maggie-content-localization/SKILL.md +41 -1
- package/bundled-skills/maggie-dash/SKILL.md +31 -0
- package/bundled-skills/maggie-deployment/SKILL.md +1 -1
- package/bundled-skills/maggie-design/SKILL.md +12 -1
- package/bundled-skills/maggie-memory/SKILL.md +7 -0
- package/bundled-skills/maggie-seo-geo/SKILL.md +75 -1
- package/bundled-skills/maggie-service-booking/SKILL.md +11 -1
- package/bundled-tools/clis/maggie_blog.py +13 -1
- package/bundled-tools/clis/maggie_browser_audit.py +78 -0
- package/bundled-tools/clis/maggie_dash.py +47 -0
- package/bundled-tools/clis/maggie_design.py +12 -0
- package/bundled-tools/clis/maggie_localization.py +42 -0
- package/bundled-tools/clis/maggie_memory.py +4 -1
- package/bundled-tools/clis/maggie_service_booking.py +3 -3
- package/bundled-tools/clis/site_audit.py +103 -8
- package/bundled-tools/runtime/browser_behavior.py +35 -0
- package/bundled-tools/runtime/browser_geometry.js +31 -0
- package/bundled-tools/runtime/localization_runner.py +113 -0
- package/bundled-tools/runtime/maggie_blog.py +66 -1
- package/bundled-tools/runtime/maggie_memory.py +7 -2
- package/bundled-tools/runtime/maggie_sitemap.py +16 -3
- package/bundled-tools/runtime/service_variants.py +152 -0
- package/bundled-tools/runtime/site_baseline.py +60 -0
- package/package.json +1 -1
- package/references/blog-translation-ingestion.md +52 -0
|
@@ -526,6 +526,18 @@ def in_place_job(project: Path, routes: list[str], force: bool = False) -> int:
|
|
|
526
526
|
project / "src" / "app" / relative / "page.tsx",
|
|
527
527
|
]
|
|
528
528
|
match = next((path for path in candidates if path.is_file()), None)
|
|
529
|
+
if not match:
|
|
530
|
+
# Dynamic catch-all routes are the source of truth for many page
|
|
531
|
+
# families; a concrete URL still needs to resolve to that route
|
|
532
|
+
# before an in-place design job can be created.
|
|
533
|
+
roots = [project / "src" / "pages", project / "src" / "app"]
|
|
534
|
+
catchalls = {"[...slug]", "[[...slug]]"}
|
|
535
|
+
for root in roots:
|
|
536
|
+
if not root.is_dir():
|
|
537
|
+
continue
|
|
538
|
+
match = next((path for path in root.rglob("*") if path.is_file() and path.stem in catchalls), None)
|
|
539
|
+
if match:
|
|
540
|
+
break
|
|
529
541
|
if not match:
|
|
530
542
|
raise ValueError(f"existing route required for in-place redesign: {route}")
|
|
531
543
|
resolved_routes.append({"route": value, "source": str(match)})
|
|
@@ -13,6 +13,27 @@ from pathlib import Path
|
|
|
13
13
|
|
|
14
14
|
sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "runtime"))
|
|
15
15
|
from content_localization import LANGUAGES, MARKETS, OPERATIONS, parse_locale, protected_field_changes, stale, valid_locale, validate_translation
|
|
16
|
+
from localization_runner import generate, process_adapter
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def generate_job(args: argparse.Namespace) -> int:
|
|
20
|
+
if not args.confirm:
|
|
21
|
+
raise ValueError("generate requires --confirm: the adapter receives source text and may incur provider costs")
|
|
22
|
+
command = json.loads(args.adapter_command)
|
|
23
|
+
adapter = process_adapter(command, args.project, args.timeout)
|
|
24
|
+
job = load(args.job)
|
|
25
|
+
errors = validate_job(args.job, args.source)["errors"]
|
|
26
|
+
if errors:
|
|
27
|
+
raise ValueError("; ".join(errors))
|
|
28
|
+
source = load(args.source)
|
|
29
|
+
contract = {key: job.get(key) for key in (
|
|
30
|
+
"jobId", "contentId", "sourceRevision", "sourceLanguage", "targetLanguage",
|
|
31
|
+
"targetLocale", "market", "operation", "generationContract")}
|
|
32
|
+
|
|
33
|
+
state = generate(source["strings"], contract, args.output, adapter, args.character_budget)
|
|
34
|
+
print(json.dumps({"status": state["status"], "strings": len(state["translations"]),
|
|
35
|
+
"output": str(args.output), "publish": False}))
|
|
36
|
+
return 0
|
|
16
37
|
|
|
17
38
|
|
|
18
39
|
def now() -> str:
|
|
@@ -186,6 +207,12 @@ def plan_job(args: argparse.Namespace) -> int:
|
|
|
186
207
|
"targetLanguage": args.target_lang,
|
|
187
208
|
"targetLocale": args.locale,
|
|
188
209
|
"operation": args.operation,
|
|
210
|
+
"generationContract": {
|
|
211
|
+
"mode": args.operation,
|
|
212
|
+
"preserveMeaning": args.operation in {"translate", "polish", "localise", "rebrand"},
|
|
213
|
+
"allowStructuralRewrite": args.operation in {"rewrite", "rebrand"},
|
|
214
|
+
"marketAdaptation": args.operation in {"localise", "rebrand"},
|
|
215
|
+
},
|
|
189
216
|
"translationGroupId": identity.get("translationGroupId", f"tg-{content_id}"),
|
|
190
217
|
"sourceRevision": args.source_revision or (source_artifact or {}).get("sourceRevision") or content.get("sourceRevision", "unknown"),
|
|
191
218
|
"status": "draft",
|
|
@@ -218,6 +245,11 @@ def validate_job(path: Path, source_path: Path | None = None, render_path: Path
|
|
|
218
245
|
if not valid_locale(job.get("targetLocale")) or parse_locale(job.get("targetLocale"))[0] != job.get("targetLanguage"):
|
|
219
246
|
errors.append("targetLocale must be a supported tag matching targetLanguage")
|
|
220
247
|
translation = job.get("translation", {})
|
|
248
|
+
generation = job.get("generationContract", {})
|
|
249
|
+
if generation.get("mode") and generation.get("mode") != job.get("operation"):
|
|
250
|
+
errors.append("generationContract mode must match operation")
|
|
251
|
+
if job.get("operation") == "rewrite" and not generation.get("allowStructuralRewrite"):
|
|
252
|
+
errors.append("rewrite requires structural rewrite permission")
|
|
221
253
|
errors.extend(validate_translation({
|
|
222
254
|
"contentId": job.get("contentId"), "lang": job.get("targetLanguage"),
|
|
223
255
|
"locale": job.get("targetLocale"), "translationGroupId": job.get("translationGroupId", f"tg-{job.get('contentId')}"),
|
|
@@ -307,6 +339,15 @@ def glossary(project: Path) -> int:
|
|
|
307
339
|
def main() -> int:
|
|
308
340
|
parser = argparse.ArgumentParser(description=__doc__)
|
|
309
341
|
sub = parser.add_subparsers(dest="command", required=True)
|
|
342
|
+
generate_parser = sub.add_parser("generate", help="Generate resumable draft strings using a trusted provider process")
|
|
343
|
+
generate_parser.add_argument("job", type=Path)
|
|
344
|
+
generate_parser.add_argument("--project", type=Path, default=Path("."))
|
|
345
|
+
generate_parser.add_argument("--source", type=Path, required=True)
|
|
346
|
+
generate_parser.add_argument("--output", type=Path, required=True)
|
|
347
|
+
generate_parser.add_argument("--adapter-command", required=True, help="JSON argv array; executed without a shell")
|
|
348
|
+
generate_parser.add_argument("--character-budget", type=int, default=8000)
|
|
349
|
+
generate_parser.add_argument("--timeout", type=int, default=120)
|
|
350
|
+
generate_parser.add_argument("--confirm", action="store_true")
|
|
310
351
|
plan_parser = sub.add_parser("plan")
|
|
311
352
|
plan_parser.add_argument("--project", type=Path, default=Path("."))
|
|
312
353
|
plan_parser.add_argument("--content", type=Path, required=True)
|
|
@@ -335,6 +376,7 @@ def main() -> int:
|
|
|
335
376
|
command = sub.add_parser(name); command.add_argument("--project", type=Path, default=Path("."))
|
|
336
377
|
args = parser.parse_args()
|
|
337
378
|
try:
|
|
379
|
+
if args.command == "generate": return generate_job(args)
|
|
338
380
|
if args.command == "plan": return plan_job(args)
|
|
339
381
|
if args.command == "preview": return preview(args.job)
|
|
340
382
|
if args.command == "validate":
|
|
@@ -13,6 +13,9 @@ from runtime import maggie_memory as memory
|
|
|
13
13
|
|
|
14
14
|
|
|
15
15
|
def output(value: object) -> None:
|
|
16
|
+
if isinstance(value, dict) and value.get("status") == "candidate" and value.get("id"):
|
|
17
|
+
value = dict(value)
|
|
18
|
+
value["notice"] = "Saved as candidate; excluded from search/context until reviewed and transitioned to active. Inspect with: maggie memory list --status candidate"
|
|
16
19
|
print(json.dumps(value, indent=2, ensure_ascii=False))
|
|
17
20
|
|
|
18
21
|
|
|
@@ -82,7 +85,7 @@ def main() -> int:
|
|
|
82
85
|
output(memory.transition(args.project, args.kind, args.item_id, args.status))
|
|
83
86
|
elif args.command in {"context", "search", "list"}:
|
|
84
87
|
query = args.query if args.command == "search" else getattr(args, "query", "")
|
|
85
|
-
result = memory.relevant(args.project, skill=getattr(args, "skill", ""), query=query)
|
|
88
|
+
result = memory.relevant(args.project, skill=getattr(args, "skill", ""), query=query, status=getattr(args, "status", None))
|
|
86
89
|
if args.command == "list" and args.kind:
|
|
87
90
|
result = {args.kind: result[args.kind]}
|
|
88
91
|
if args.command == "list" and args.status:
|
|
@@ -793,9 +793,9 @@ def cmd_match_pages_review(args):
|
|
|
793
793
|
out=project/".maggie"/"booking"/"page-matches.json"; out.parent.mkdir(parents=True,exist_ok=True); (project/"docs").mkdir(parents=True,exist_ok=True); payload={"matchedAt":NOW(),"requiresManualSelection":True,"candidateServices":report,"selected":0,"conflicts":conflicts}; out.write_text(json.dumps(payload,indent=2,ensure_ascii=False)+"\n",encoding="utf-8"); (project/"docs"/"service-page-matches.json").write_text(json.dumps(payload,indent=2,ensure_ascii=False)+"\n",encoding="utf-8"); print(json.dumps({"status":"conflict-review","pagesScanned":len(candidates),"servicesScanned":len(report),"selected":0,"conflicts":conflicts,"report":str(out)},indent=2)); return 1
|
|
794
794
|
for service,path_value in selected:
|
|
795
795
|
relation={"path":path_value,"role":"canonical","matchMethod":"manual-selection","confidence":1.0}; service["pages"]=[p for p in service.get("pages",[]) if p.get("path")!=path_value]+[relation]
|
|
796
|
-
|
|
797
|
-
|
|
798
|
-
|
|
796
|
+
# Reviewed selections must not erase existing supporting relations. They
|
|
797
|
+
# may have been imported from a route table or authored intentionally even
|
|
798
|
+
# when the current page scanner cannot see their source file.
|
|
799
799
|
out=project/".maggie"/"booking"/"page-matches.json"; payload={"matchedAt":NOW(),"requiresManualSelection":True,"candidateServices":report,"selected":len(selected),"conflicts":[]}; out.parent.mkdir(parents=True,exist_ok=True); (project/"docs").mkdir(parents=True,exist_ok=True); out.write_text(json.dumps(payload,indent=2,ensure_ascii=False)+"\n",encoding="utf-8"); (project/"docs"/"service-page-matches.json").write_text(json.dumps(payload,indent=2,ensure_ascii=False)+"\n",encoding="utf-8"); save(project,data); print(json.dumps({"status":"review-required","pagesScanned":len(candidates),"servicesScanned":len(report),"selected":len(selected),"conflicts":[],"report":str(out)},indent=2)); return 0
|
|
800
800
|
def cmd_run(args):
|
|
801
801
|
project=root(args); job={"workflow":"maggie-service-booking","phase":"created","source":args.source,"provider":args.provider,"startedAt":NOW(),"history":[]}
|
|
@@ -4,6 +4,7 @@
|
|
|
4
4
|
from __future__ import annotations
|
|
5
5
|
|
|
6
6
|
import argparse
|
|
7
|
+
import hashlib
|
|
7
8
|
import json
|
|
8
9
|
import re
|
|
9
10
|
import sys
|
|
@@ -15,6 +16,7 @@ from urllib.request import Request, urlopen
|
|
|
15
16
|
|
|
16
17
|
sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "runtime"))
|
|
17
18
|
from content_localization import MARKETS, parse_locale, valid_locale # noqa: E402
|
|
19
|
+
import site_baseline
|
|
18
20
|
|
|
19
21
|
|
|
20
22
|
class PageParser(HTMLParser):
|
|
@@ -32,13 +34,27 @@ class PageParser(HTMLParser):
|
|
|
32
34
|
self._jsonld = None
|
|
33
35
|
self.images = []
|
|
34
36
|
self.hreflang = []
|
|
37
|
+
self.robots_directives = []
|
|
38
|
+
self.structure = []
|
|
39
|
+
self.visible_text = []
|
|
40
|
+
self.hidden_depth = 0
|
|
41
|
+
self.templates = set()
|
|
35
42
|
|
|
36
43
|
def handle_starttag(self, tag, attrs):
|
|
37
44
|
data = dict(attrs)
|
|
45
|
+
if data.get("data-template"):
|
|
46
|
+
self.templates.add(data["data-template"])
|
|
47
|
+
if tag in {"script", "style", "noscript"}:
|
|
48
|
+
self.hidden_depth += 1
|
|
49
|
+
if not self.hidden_depth:
|
|
50
|
+
self.structure.append(["start", tag, {key: data[key] for key in
|
|
51
|
+
("class", "id", "role", "href", "src", "data-template", "data-section", "data-i18n") if key in data}])
|
|
38
52
|
if tag == "html":
|
|
39
53
|
self.lang = data.get("lang", "")
|
|
40
54
|
if tag == "meta" and data.get("name"):
|
|
41
55
|
self.meta[data["name"].lower()] = data.get("content", "")
|
|
56
|
+
if data["name"].lower() == "robots":
|
|
57
|
+
self.robots_directives.append(data.get("content", "").lower())
|
|
42
58
|
if tag == "meta" and data.get("property"):
|
|
43
59
|
self.meta[data["property"].lower()] = data.get("content", "")
|
|
44
60
|
if tag == "title":
|
|
@@ -58,6 +74,10 @@ class PageParser(HTMLParser):
|
|
|
58
74
|
self.images.append({"src": data.get("src", ""), "alt": data.get("alt")})
|
|
59
75
|
|
|
60
76
|
def handle_endtag(self, tag):
|
|
77
|
+
if tag in {"script", "style", "noscript"}:
|
|
78
|
+
self.hidden_depth = max(0, self.hidden_depth - 1)
|
|
79
|
+
elif not self.hidden_depth:
|
|
80
|
+
self.structure.append(["end", tag])
|
|
61
81
|
if tag == "title":
|
|
62
82
|
self.in_title = False
|
|
63
83
|
if tag == "script" and self._jsonld is not None:
|
|
@@ -69,23 +89,35 @@ class PageParser(HTMLParser):
|
|
|
69
89
|
self._jsonld = None
|
|
70
90
|
|
|
71
91
|
def handle_data(self, data):
|
|
92
|
+
if not self.hidden_depth and data.strip():
|
|
93
|
+
self.visible_text.append(" ".join(data.split()))
|
|
72
94
|
if self.in_title:
|
|
73
95
|
self.title += data.strip()
|
|
74
96
|
if self._jsonld is not None:
|
|
75
97
|
self._jsonld.append(data)
|
|
76
98
|
|
|
77
99
|
|
|
78
|
-
def fetch(url: str) -> tuple[int, str, str]:
|
|
100
|
+
def fetch(url: str, evidence: dict | None = None) -> tuple[int, str, str]:
|
|
79
101
|
request = Request(url, headers={"User-Agent": "AI-CMO-Skills-Audit/0.1"})
|
|
80
102
|
with urlopen(request, timeout=15) as response:
|
|
81
|
-
|
|
103
|
+
raw = response.read(2_000_001)
|
|
104
|
+
if len(raw) > 2_000_000:
|
|
105
|
+
raise ValueError("response exceeds audit size limit")
|
|
106
|
+
body = raw.decode(response.headers.get_content_charset() or "utf-8", "replace")
|
|
107
|
+
if evidence is not None:
|
|
108
|
+
evidence.update({"finalUrl": response.geturl(),
|
|
109
|
+
"xRobotsTag": response.headers.get_all("X-Robots-Tag", [])})
|
|
82
110
|
return response.status, response.headers.get_content_type(), body
|
|
83
111
|
|
|
84
112
|
|
|
85
|
-
def audit_page(url: str, html: str, status: int, content_type: str, expected_languages: set[str] | None = None) -> dict:
|
|
113
|
+
def audit_page(url: str, html: str, status: int, content_type: str, expected_languages: set[str] | None = None, response_evidence: dict | None = None) -> dict:
|
|
86
114
|
page = PageParser()
|
|
87
115
|
page.feed(html)
|
|
88
116
|
canonical = urljoin(url, page.canonical) if page.canonical else ""
|
|
117
|
+
response_evidence = response_evidence or {"finalUrl": url, "xRobotsTag": []}
|
|
118
|
+
robots = page.robots_directives + [value.lower() for value in response_evidence["xRobotsTag"]]
|
|
119
|
+
robots_tokens = [token.strip() for directive in robots for token in directive.split(",")]
|
|
120
|
+
robots_tokens = [token.rsplit(":", 1)[-1].strip() for token in robots_tokens]
|
|
89
121
|
return {
|
|
90
122
|
"url": url,
|
|
91
123
|
"status": status,
|
|
@@ -104,10 +136,45 @@ def audit_page(url: str, html: str, status: int, content_type: str, expected_lan
|
|
|
104
136
|
"canonical_value": canonical,
|
|
105
137
|
"image_count": len(page.images),
|
|
106
138
|
"language": parse_locale(page.lang)[0],
|
|
139
|
+
"declaredLocale": page.lang,
|
|
140
|
+
"template": next(iter(page.templates)) if len(page.templates) == 1 else "unknown",
|
|
141
|
+
"evidence": {"structural": "captured", "rendered": "not_run", "behavioral": "not_run"},
|
|
107
142
|
"hreflang": page.hreflang,
|
|
143
|
+
"contract": {
|
|
144
|
+
"response": response_evidence,
|
|
145
|
+
"title": page.title, "meta": page.meta, "canonical": canonical,
|
|
146
|
+
"lang": page.lang, "hreflang": page.hreflang, "robots": robots,
|
|
147
|
+
"jsonld": page.jsonld_values, "images": page.images,
|
|
148
|
+
"structureHash": hashlib.sha256(json.dumps(page.structure, sort_keys=True).encode()).hexdigest(),
|
|
149
|
+
"textHash": hashlib.sha256(" ".join(page.visible_text).encode()).hexdigest(),
|
|
150
|
+
},
|
|
151
|
+
"robots": not robots or not any(token in {"noindex", "none", "nofollow"} for token in robots_tokens),
|
|
152
|
+
"robots_directives": robots,
|
|
153
|
+
"robots_conflict": len({token for token in robots_tokens if token in {"index", "noindex", "follow", "nofollow", "none"}} & {"index", "noindex"}) > 1 or len({token for token in robots_tokens if token in {"follow", "nofollow", "none"}} & {"follow", "nofollow"}) > 1,
|
|
108
154
|
}
|
|
109
155
|
|
|
110
156
|
|
|
157
|
+
def summarize_crawl(pages: list[dict]) -> dict:
|
|
158
|
+
"""Count captured evidence, preserving regional/script locale distinctions."""
|
|
159
|
+
groups = {}
|
|
160
|
+
for page in pages:
|
|
161
|
+
locale = page.get("declaredLocale") or "unknown"
|
|
162
|
+
template = page.get("template") or "unknown"
|
|
163
|
+
key = (locale, template)
|
|
164
|
+
group = groups.setdefault(key, {"locale": locale, "template": template,
|
|
165
|
+
"total": 0, "passed": 0, "failed": 0, "failedUrls": [],
|
|
166
|
+
"evidence": {"structural": "captured", "rendered": "not_run", "behavioral": "not_run"}})
|
|
167
|
+
group["total"] += 1
|
|
168
|
+
if page.get("passed") is True:
|
|
169
|
+
group["passed"] += 1
|
|
170
|
+
else:
|
|
171
|
+
group["failed"] += 1
|
|
172
|
+
group["failedUrls"].append(page["url"])
|
|
173
|
+
for group in groups.values():
|
|
174
|
+
group["failedUrls"].sort()
|
|
175
|
+
return {"byLocaleTemplate": [groups[key] for key in sorted(groups)]}
|
|
176
|
+
|
|
177
|
+
|
|
111
178
|
def language_set(value: str | None) -> set[str]:
|
|
112
179
|
if not value:
|
|
113
180
|
return set()
|
|
@@ -170,7 +237,17 @@ def main() -> int:
|
|
|
170
237
|
parser.add_argument("--markets", help="comma-separated markets: global,uk,us")
|
|
171
238
|
parser.add_argument("--check-hreflang", action="store_true")
|
|
172
239
|
parser.add_argument("--check-translation-completeness", action="store_true")
|
|
240
|
+
baseline_args = parser.add_mutually_exclusive_group()
|
|
241
|
+
baseline_args.add_argument("--save-baseline", type=Path, help="create a new reviewed contract from a passing complete crawl")
|
|
242
|
+
baseline_args.add_argument("--baseline", type=Path, help="fail on differences from a reviewed contract")
|
|
243
|
+
parser.add_argument("--reviewer", help="required for --save-baseline")
|
|
173
244
|
args = parser.parse_args()
|
|
245
|
+
if args.max_pages < 1:
|
|
246
|
+
parser.error("max-pages must be positive")
|
|
247
|
+
if (args.save_baseline or args.baseline) and not args.crawl:
|
|
248
|
+
parser.error("baseline operations require --crawl")
|
|
249
|
+
if args.save_baseline and not (args.reviewer or "").strip():
|
|
250
|
+
parser.error("save-baseline requires --reviewer")
|
|
174
251
|
try:
|
|
175
252
|
expected_languages = language_set(args.languages)
|
|
176
253
|
markets = {item.strip() for item in (args.markets or "").split(",") if item.strip()}
|
|
@@ -197,6 +274,9 @@ def main() -> int:
|
|
|
197
274
|
checks["jsonld"] = {"ok": page.jsonld > 0 and all(item is not None for item in page.jsonld_values), "count": page.jsonld}
|
|
198
275
|
checks["entity_jsonld"] = {"ok": any(isinstance(item, dict) and item.get("@type") and (item.get("url") or item.get("@id")) for item in page.jsonld_values), "count": page.jsonld}
|
|
199
276
|
checks["crawlable_links"] = {"ok": page.anchors > 0, "count": page.anchors}
|
|
277
|
+
robots_tokens = [token.strip() for directive in page.robots_directives for token in directive.split(",")]
|
|
278
|
+
checks["robots_directive"] = {"ok": not page.robots_directives or not any(token in {"noindex", "none", "nofollow"} for token in robots_tokens), "directives": page.robots_directives}
|
|
279
|
+
checks["robots_conflict"] = {"ok": not (len({token for token in robots_tokens if token in {"index", "noindex"}}) > 1 or len({token for token in robots_tokens if token in {"follow", "nofollow"}}) > 1), "directives": page.robots_directives}
|
|
200
280
|
if args.check_hreflang or args.check_translation_completeness:
|
|
201
281
|
checks["hreflang"] = hreflang_check(base, page, expected_languages) if args.check_hreflang else {"ok": True, "links": page.hreflang}
|
|
202
282
|
if args.check_translation_completeness and expected_languages:
|
|
@@ -221,11 +301,14 @@ def main() -> int:
|
|
|
221
301
|
if args.crawl:
|
|
222
302
|
try:
|
|
223
303
|
urls, sitemap_violations = sitemap_urls(base)
|
|
304
|
+
crawl["discovered_url_count"] = len(urls)
|
|
305
|
+
crawl["complete"] = len(urls) <= args.max_pages and not sitemap_violations
|
|
224
306
|
urls = urls[: args.max_pages]
|
|
225
307
|
for page_url in urls:
|
|
226
308
|
try:
|
|
227
|
-
|
|
228
|
-
|
|
309
|
+
response_evidence = {}
|
|
310
|
+
status, content_type, body = fetch(page_url, response_evidence)
|
|
311
|
+
page_checks = audit_page(page_url, body, status, content_type, expected_languages, response_evidence)
|
|
229
312
|
if args.check_hreflang or args.check_translation_completeness:
|
|
230
313
|
parsed = PageParser(); parsed.feed(body)
|
|
231
314
|
page_checks["hreflang_check"] = hreflang_check(page_url, parsed, expected_languages) if args.check_hreflang else {"ok": True, "links": parsed.hreflang}
|
|
@@ -233,18 +316,30 @@ def main() -> int:
|
|
|
233
316
|
page_checks["translation_completeness"] = {"ok": all(lang in {item["lang"] for item in parsed.hreflang} for lang in expected_languages), "expected": sorted(expected_languages)}
|
|
234
317
|
page_checks["passed"] = all(
|
|
235
318
|
page_checks[key]
|
|
236
|
-
for key in ("ok", "title", "h1", "canonical", "description", "locale", "open_graph", "twitter_card", "jsonld", "entity_jsonld", "images")
|
|
237
|
-
) and (not args.check_hreflang or page_checks["hreflang_check"]["ok"]) and (not args.check_translation_completeness or page_checks.get("translation_completeness", {}).get("ok", False))
|
|
319
|
+
for key in ("ok", "title", "h1", "canonical", "description", "locale", "open_graph", "twitter_card", "jsonld", "entity_jsonld", "images", "robots")
|
|
320
|
+
) and not page_checks["robots_conflict"] and (not args.check_hreflang or page_checks["hreflang_check"]["ok"]) and (not args.check_translation_completeness or page_checks.get("translation_completeness", {}).get("ok", False))
|
|
238
321
|
except Exception as exc:
|
|
239
322
|
page_checks = {"url": page_url, "passed": False, "error": type(exc).__name__}
|
|
240
323
|
crawl["pages"].append(page_checks)
|
|
241
324
|
crawl["sitemap_loc_violations"] = sitemap_violations
|
|
242
|
-
crawl["passed"] =
|
|
325
|
+
crawl["passed"] = crawl["complete"] and bool(urls) and len(crawl["pages"]) == len(urls) and all(item["passed"] for item in crawl["pages"])
|
|
243
326
|
crawl["url_count"] = len(urls)
|
|
244
327
|
except Exception as exc:
|
|
245
328
|
crawl = {"enabled": True, "passed": False, "error": type(exc).__name__, "pages": []}
|
|
246
329
|
|
|
247
330
|
result = {"url": base, "environment": args.environment, "checks": checks, "crawl": crawl, "passed": all(v.get("ok", False) for v in checks.values()) and crawl["passed"]}
|
|
331
|
+
result["evidence"] = {"structural": "captured", "rendered": "not_run", "behavioral": "not_run"}
|
|
332
|
+
crawl["summary"] = summarize_crawl(crawl["pages"])
|
|
333
|
+
try:
|
|
334
|
+
if args.baseline:
|
|
335
|
+
result["baseline"] = site_baseline.compare(json.loads(args.baseline.read_text()), result)
|
|
336
|
+
result["passed"] = result["passed"] and result["baseline"]["passed"]
|
|
337
|
+
if args.save_baseline:
|
|
338
|
+
site_baseline.save(args.save_baseline, site_baseline.snapshot(result, args.reviewer))
|
|
339
|
+
result["baseline"] = {"passed": True, "saved": str(args.save_baseline)}
|
|
340
|
+
except (OSError, ValueError, TypeError) as exc:
|
|
341
|
+
result["baseline"] = {"passed": False, "error": str(exc)}
|
|
342
|
+
result["passed"] = False
|
|
248
343
|
if args.output:
|
|
249
344
|
output = Path(args.output).resolve()
|
|
250
345
|
output.parent.mkdir(parents=True, exist_ok=True)
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
"""Compare actual browser samples against explicit route expectations."""
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
def validate_samples(samples: list[dict], selectors: list[str], sticky: list[str]) -> dict:
|
|
5
|
+
errors = []
|
|
6
|
+
if len(samples) < 2:
|
|
7
|
+
return {"passed": False, "errors": ["at least two scroll samples required"]}
|
|
8
|
+
first = samples[0]
|
|
9
|
+
for sample in samples:
|
|
10
|
+
if sample.get("viewport") != first.get("viewport") or sample.get("url") != first.get("url"):
|
|
11
|
+
errors.append("samples must share URL and viewport")
|
|
12
|
+
if sample.get("horizontalOverflow"):
|
|
13
|
+
errors.append("horizontal overflow")
|
|
14
|
+
elements = {element["selector"]: element for element in sample.get("elements", [])}
|
|
15
|
+
for selector in selectors:
|
|
16
|
+
element = elements.get(selector, {})
|
|
17
|
+
if not element.get("found") or not element.get("visible"):
|
|
18
|
+
errors.append(f"required element missing or hidden: {selector}")
|
|
19
|
+
if samples[-1].get("scroll", {}).get("y", 0) <= first.get("scroll", {}).get("y", 0):
|
|
20
|
+
errors.append("scroll did not advance; behavior unverified")
|
|
21
|
+
for selector in sticky:
|
|
22
|
+
for sample in samples[1:]:
|
|
23
|
+
element = next((item for item in sample.get("elements", []) if item.get("selector") == selector), {})
|
|
24
|
+
if element.get("position") not in {"sticky", "fixed"}:
|
|
25
|
+
errors.append(f"expected sticky/fixed positioning: {selector}")
|
|
26
|
+
continue
|
|
27
|
+
top = element.get("rect", {}).get("top")
|
|
28
|
+
try:
|
|
29
|
+
inset = float(element.get("insetTop", "auto").removesuffix("px"))
|
|
30
|
+
except (ValueError, AttributeError):
|
|
31
|
+
errors.append(f"sticky top inset unmeasurable: {selector}")
|
|
32
|
+
continue
|
|
33
|
+
if not isinstance(top, (int, float)) or abs(top - inset) > 2:
|
|
34
|
+
errors.append(f"sticky element left expected top inset: {selector}")
|
|
35
|
+
return {"passed": not errors, "errors": sorted(set(errors))}
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
// Read-only browser measurements. The caller supplies selectors and scrolls
|
|
2
|
+
// between samples; never infer scrolling behavior from position:sticky alone.
|
|
3
|
+
((selectors = window.__maggieAuditSelectors || []) => {
|
|
4
|
+
const visible = element => {
|
|
5
|
+
const rect = element.getBoundingClientRect();
|
|
6
|
+
if (!rect.width || !rect.height) return false;
|
|
7
|
+
for (let node = element; node; node = node.parentElement) {
|
|
8
|
+
const style = getComputedStyle(node);
|
|
9
|
+
if (style.display === 'none' || style.visibility === 'hidden' || Number(style.opacity) === 0) return false;
|
|
10
|
+
}
|
|
11
|
+
return true;
|
|
12
|
+
};
|
|
13
|
+
return {
|
|
14
|
+
url: location.href,
|
|
15
|
+
viewport: {width: innerWidth, height: innerHeight},
|
|
16
|
+
scroll: {x: scrollX, y: scrollY, maxY: Math.max(0, document.documentElement.scrollHeight - innerHeight)},
|
|
17
|
+
horizontalOverflow: document.documentElement.scrollWidth > innerWidth + 1,
|
|
18
|
+
elements: selectors.map(selector => {
|
|
19
|
+
const element = document.querySelector(selector);
|
|
20
|
+
if (!element) return {selector, found: false, visible: false};
|
|
21
|
+
const rect = element.getBoundingClientRect();
|
|
22
|
+
const style = getComputedStyle(element);
|
|
23
|
+
const parent = element.parentElement?.getBoundingClientRect();
|
|
24
|
+
return {selector, found: true, visible: visible(element),
|
|
25
|
+
rect: {top: rect.top, bottom: rect.bottom, left: rect.left, right: rect.right, width: rect.width, height: rect.height},
|
|
26
|
+
position: style.position, insetTop: style.top,
|
|
27
|
+
parentRect: parent ? {top: parent.top, bottom: parent.bottom, height: parent.height} : null,
|
|
28
|
+
hiddenDescendants: [...element.querySelectorAll('a,button,input,select')].filter(child => !visible(child)).length};
|
|
29
|
+
})
|
|
30
|
+
};
|
|
31
|
+
})()
|
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
"""Resumable translation batches; provider calls are supplied by the adapter."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import hashlib
|
|
5
|
+
import json
|
|
6
|
+
import os
|
|
7
|
+
import subprocess
|
|
8
|
+
import tempfile
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def process_adapter(command: list[str], project: Path, timeout: int = 120):
|
|
13
|
+
if not isinstance(command, list) or not command or any(not isinstance(part, str) or not part for part in command):
|
|
14
|
+
raise ValueError("adapter-command must be a nonempty JSON argv array")
|
|
15
|
+
if timeout < 1:
|
|
16
|
+
raise ValueError("timeout must be positive")
|
|
17
|
+
def adapter(batch, contract):
|
|
18
|
+
try:
|
|
19
|
+
result = subprocess.run(command, input=json.dumps({
|
|
20
|
+
"schemaVersion": "maggie-provider-request.v1", "strings": batch,
|
|
21
|
+
"contract": contract}, ensure_ascii=False), text=True, capture_output=True,
|
|
22
|
+
timeout=timeout, check=False, cwd=project.resolve())
|
|
23
|
+
except (OSError, subprocess.TimeoutExpired):
|
|
24
|
+
raise OSError("provider adapter could not execute or timed out; retry to resume") from None
|
|
25
|
+
if result.returncode:
|
|
26
|
+
raise OSError("provider adapter failed; retry to resume (provider logs suppressed)")
|
|
27
|
+
return result.stdout
|
|
28
|
+
return adapter
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def checkpoint(path: Path, state: dict) -> None:
|
|
32
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
33
|
+
descriptor, name = tempfile.mkstemp(prefix=".translation-", dir=path.parent)
|
|
34
|
+
try:
|
|
35
|
+
with os.fdopen(descriptor, "w", encoding="utf-8") as stream:
|
|
36
|
+
json.dump(state, stream, ensure_ascii=False, indent=2)
|
|
37
|
+
stream.flush()
|
|
38
|
+
os.fsync(stream.fileno())
|
|
39
|
+
os.replace(name, path)
|
|
40
|
+
finally:
|
|
41
|
+
if os.path.exists(name):
|
|
42
|
+
os.unlink(name)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def generate(strings: list[dict], contract: dict, output: Path, adapter,
|
|
46
|
+
character_budget: int = 8000) -> dict:
|
|
47
|
+
"""Adapter accepts (batch, contract) and returns an exact ID-to-text map.
|
|
48
|
+
|
|
49
|
+
Invalid/truncated output splits a batch. Provider exceptions propagate;
|
|
50
|
+
successfully checkpointed batches are reusable on the next invocation.
|
|
51
|
+
"""
|
|
52
|
+
if type(character_budget) is not int or character_budget < 1:
|
|
53
|
+
raise ValueError("character_budget must be positive")
|
|
54
|
+
if not strings or any(not isinstance(item, dict) or not isinstance(item.get("id"), str)
|
|
55
|
+
or not item["id"] or not isinstance(item.get("text"), str)
|
|
56
|
+
or not item["text"].strip() for item in strings):
|
|
57
|
+
raise ValueError("source strings require nonempty IDs and text")
|
|
58
|
+
if len({item["id"] for item in strings}) != len(strings):
|
|
59
|
+
raise ValueError("duplicate source IDs")
|
|
60
|
+
identity = hashlib.sha256(json.dumps({"strings": strings, "contract": contract},
|
|
61
|
+
sort_keys=True, ensure_ascii=False).encode()).hexdigest()
|
|
62
|
+
state = {"schemaVersion": "maggie-generation.v1", "identity": identity,
|
|
63
|
+
"translations": {}, "status": "pending", "publish": False}
|
|
64
|
+
if output.exists():
|
|
65
|
+
state = json.loads(output.read_text(encoding="utf-8"))
|
|
66
|
+
if not isinstance(state, dict) or state.get("schemaVersion") != "maggie-generation.v1" or state.get("publish") is not False:
|
|
67
|
+
raise ValueError("invalid generation checkpoint")
|
|
68
|
+
if state.get("identity") != identity:
|
|
69
|
+
raise ValueError("source or generation contract changed; choose a new output")
|
|
70
|
+
translations = state.get("translations")
|
|
71
|
+
if not isinstance(translations, dict) or any(key not in {s["id"] for s in strings}
|
|
72
|
+
or not isinstance(value, str) or not value.strip() for key, value in translations.items()):
|
|
73
|
+
raise ValueError("invalid generation checkpoint")
|
|
74
|
+
|
|
75
|
+
def run(batch):
|
|
76
|
+
try:
|
|
77
|
+
result = adapter(batch, contract)
|
|
78
|
+
if isinstance(result, str):
|
|
79
|
+
result = json.loads(result)
|
|
80
|
+
if not isinstance(result, dict) or set(result) != {item["id"] for item in batch}:
|
|
81
|
+
raise ValueError("provider returned missing or unexpected IDs")
|
|
82
|
+
if any(not isinstance(value, str) or not value.strip() for value in result.values()):
|
|
83
|
+
raise ValueError("provider returned empty translations")
|
|
84
|
+
except (ValueError, json.JSONDecodeError):
|
|
85
|
+
if len(batch) == 1:
|
|
86
|
+
raise ValueError("provider output invalid for single source string") from None
|
|
87
|
+
midpoint = len(batch) // 2
|
|
88
|
+
run(batch[:midpoint])
|
|
89
|
+
run(batch[midpoint:])
|
|
90
|
+
return
|
|
91
|
+
state["translations"].update(result)
|
|
92
|
+
checkpoint(output, state)
|
|
93
|
+
|
|
94
|
+
pending = [item for item in strings if item["id"] not in state["translations"]]
|
|
95
|
+
state["status"] = "running"
|
|
96
|
+
checkpoint(output, state)
|
|
97
|
+
try:
|
|
98
|
+
batch, size = [], 0
|
|
99
|
+
for item in pending:
|
|
100
|
+
if batch and size + len(item["text"]) > character_budget:
|
|
101
|
+
run(batch)
|
|
102
|
+
batch, size = [], 0
|
|
103
|
+
batch.append(item)
|
|
104
|
+
size += len(item["text"])
|
|
105
|
+
if batch:
|
|
106
|
+
run(batch)
|
|
107
|
+
except Exception:
|
|
108
|
+
state["status"] = "failed"
|
|
109
|
+
checkpoint(output, state)
|
|
110
|
+
raise
|
|
111
|
+
state["status"] = "generated"
|
|
112
|
+
checkpoint(output, state)
|
|
113
|
+
return state
|
|
@@ -8,6 +8,8 @@ import re
|
|
|
8
8
|
import shutil
|
|
9
9
|
from datetime import datetime, timezone
|
|
10
10
|
from pathlib import Path
|
|
11
|
+
from localization_runner import checkpoint, generate
|
|
12
|
+
from content_localization import valid_locale
|
|
11
13
|
|
|
12
14
|
STATUSES = ("draft", "review", "approved", "published", "archived")
|
|
13
15
|
DEFAULTS = {
|
|
@@ -20,6 +22,8 @@ DEFAULTS = {
|
|
|
20
22
|
"pullIntervalMinutes": 120,
|
|
21
23
|
"maxPerRun": 1,
|
|
22
24
|
"fallbackImageMode": "gradient",
|
|
25
|
+
"autoTranslateEnabled": False,
|
|
26
|
+
"translationLocales": [],
|
|
23
27
|
}
|
|
24
28
|
|
|
25
29
|
|
|
@@ -65,7 +69,62 @@ class BlogStore:
|
|
|
65
69
|
return json.loads(self.posts_path.read_text(encoding="utf-8"))
|
|
66
70
|
|
|
67
71
|
def _write(self, path: Path, value: object) -> None:
|
|
68
|
-
path
|
|
72
|
+
checkpoint(path, value)
|
|
73
|
+
|
|
74
|
+
def reconcile_translations(self) -> list[dict]:
|
|
75
|
+
"""Recover missing scheduling intents from persisted source, without re-pull."""
|
|
76
|
+
settings = self.settings()
|
|
77
|
+
if not settings.get("autoTranslateEnabled"):
|
|
78
|
+
return []
|
|
79
|
+
locales = settings.get("translationLocales", [])
|
|
80
|
+
if not isinstance(locales, list) or not locales or any(not isinstance(locale, str) or not valid_locale(locale) for locale in locales):
|
|
81
|
+
raise ValueError("auto translation requires supported translationLocales")
|
|
82
|
+
tasks = []
|
|
83
|
+
for post in self.posts():
|
|
84
|
+
for locale in sorted(set(locales)):
|
|
85
|
+
contract = {"projectId": post["projectId"], "contentId": post["contentId"],
|
|
86
|
+
"sourceRevision": checksum({key: post[key] for key in
|
|
87
|
+
("title", "excerpt", "body", "topics", "source")}
|
|
88
|
+
| {"images": post.get("images", [])}),
|
|
89
|
+
"targetLocale": locale, "operation": "translate"}
|
|
90
|
+
identity = checksum(contract).split(":")[1]
|
|
91
|
+
path = self.root / "translations" / (identity + ".json")
|
|
92
|
+
if not path.exists():
|
|
93
|
+
self._write(path, {"id": identity, "contract": contract, "status": "pending",
|
|
94
|
+
"attempts": 0, "isIndexable": False})
|
|
95
|
+
tasks.append(json.loads(path.read_text(encoding="utf-8")))
|
|
96
|
+
return tasks
|
|
97
|
+
|
|
98
|
+
def translate_pending(self, adapter, character_budget: int = 8000) -> dict:
|
|
99
|
+
"""Single-worker draft translation; every caller uses reconciliation first."""
|
|
100
|
+
tasks = self.reconcile_translations()
|
|
101
|
+
posts = {post["contentId"]: post for post in self.posts()}
|
|
102
|
+
for task in tasks:
|
|
103
|
+
if task["status"] == "succeeded":
|
|
104
|
+
continue
|
|
105
|
+
post = posts[task["contract"]["contentId"]]
|
|
106
|
+
strings = [{"id": key, "text": text} for key, text in (
|
|
107
|
+
("title", post["title"]), ("excerpt", post["excerpt"]),
|
|
108
|
+
("body", post["body"].get("value", ""))) if isinstance(text, str) and text.strip()]
|
|
109
|
+
strings.extend({"id": "topic:" + str(index), "text": topic["label"]}
|
|
110
|
+
for index, topic in enumerate(post["topics"]))
|
|
111
|
+
strings.extend({"id": "alt:" + str(index), "text": image["alt"]}
|
|
112
|
+
for index, image in enumerate(post.get("images", []))
|
|
113
|
+
if isinstance(image.get("alt"), str) and image["alt"].strip())
|
|
114
|
+
path = self.root / "translations" / (task["id"] + ".json")
|
|
115
|
+
task.update(status="running", attempts=task["attempts"] + 1)
|
|
116
|
+
self._write(path, task)
|
|
117
|
+
try:
|
|
118
|
+
result = generate(strings, task["contract"],
|
|
119
|
+
self.root / "translations" / "checkpoints" / (task["id"] + ".json"),
|
|
120
|
+
adapter, character_budget)
|
|
121
|
+
task.update(status="succeeded", draft=result["translations"], isIndexable=False)
|
|
122
|
+
task.pop("errorCategory", None)
|
|
123
|
+
except Exception as error:
|
|
124
|
+
task.update(status="failed", errorCategory=type(error).__name__)
|
|
125
|
+
self._write(path, task)
|
|
126
|
+
return {"status": "partial" if any(task["status"] != "succeeded" for task in tasks) else "completed",
|
|
127
|
+
"tasks": [{key: task[key] for key in ("id", "status", "attempts")} for task in tasks]}
|
|
69
128
|
|
|
70
129
|
def _backup(self) -> None:
|
|
71
130
|
if self.posts_path.exists():
|
|
@@ -91,6 +150,8 @@ class BlogStore:
|
|
|
91
150
|
"slug": old["slug"] if old else slugify(raw_slug),
|
|
92
151
|
"title": title,
|
|
93
152
|
"excerpt": str(item.get("excerpt") or ""),
|
|
153
|
+
"images": [{"src": image.get("src", ""), "alt": image.get("alt", "")}
|
|
154
|
+
for image in item.get("images", []) if isinstance(image, dict)],
|
|
94
155
|
"body": item.get("body") if isinstance(item.get("body"), dict) else {"format": str(item.get("format") or "markdown"), "value": str(item.get("body") or item.get("content") or "")},
|
|
95
156
|
"topics": [{"slug": slugify(str(topic)), "label": str(topic)} for topic in item.get("topics", item.get("keywords", []))],
|
|
96
157
|
# Provider input can suggest a status, but cannot bypass the
|
|
@@ -111,8 +172,12 @@ class BlogStore:
|
|
|
111
172
|
slugs = [post["slug"] for post in result]
|
|
112
173
|
if len(slugs) != len(set(slugs)): raise ValueError("slug collision detected; provide unique slugs")
|
|
113
174
|
self._backup(); self._write(self.posts_path, result)
|
|
175
|
+
translation_tasks = self.reconcile_translations()
|
|
114
176
|
runs = json.loads(self.runs_path.read_text(encoding="utf-8")) if self.runs_path.exists() else []
|
|
115
177
|
run = {"id": "pull:" + hashlib.sha256((now() + provider).encode()).hexdigest()[:12], "provider": provider, "status": "completed", "delivered": len(payload), "changed": changed, "finishedAt": now()}
|
|
178
|
+
run["translationTasks"] = len(translation_tasks)
|
|
179
|
+
if any(task["status"] != "succeeded" for task in translation_tasks):
|
|
180
|
+
run["status"] = "partial"
|
|
116
181
|
runs.append(run); self._write(self.runs_path, runs)
|
|
117
182
|
return run
|
|
118
183
|
|