@topy-ai/maggie 0.7.9 → 0.7.11

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -328,7 +328,7 @@ PHASES = [
328
328
  ],
329
329
  },
330
330
  ]
331
- DESIGN_JOB_PHASES = ("created", "preflight", "planned", "validated", "ready", "failed")
331
+ DESIGN_JOB_PHASES = ("created", "preflight", "planned", "validated", "ready", "done", "failed")
332
332
 
333
333
 
334
334
  def existing_evidence(project: Path, target: dict[str, str]) -> list[str]:
@@ -506,7 +506,32 @@ def design_resume(project: Path, job_id: str) -> int:
506
506
  return run_design_job(job["urls"], project, job["clone_run"], force=True)[0]
507
507
 
508
508
 
509
- def in_place_job(project: Path, routes: list[str], force: bool = False) -> int:
509
+ def _step_records(names: list[str]) -> list[dict[str, object]]:
510
+ return [{"name": name, "status": "pending", "evidence": None} for name in names]
511
+
512
+
513
+ def design_step(project: Path, job_id: str, step: str, evidence: str) -> int:
514
+ """Record one piece of implementation evidence and make progress durable."""
515
+ path = design_job_path(project.resolve(), job_id)
516
+ if not path.exists():
517
+ raise ValueError(f"design job not found: {path}")
518
+ job = json.loads(path.read_text(encoding="utf-8"))
519
+ steps = job.get("steps", [])
520
+ target = next((item for item in steps if item.get("name") == step), None)
521
+ if target is None:
522
+ raise ValueError(f"unknown design step for {job_id}: {step}")
523
+ evidence_path = Path(evidence).expanduser().resolve()
524
+ if not evidence_path.exists():
525
+ raise ValueError(f"step evidence does not exist: {evidence_path}")
526
+ target.update({"status": "done", "evidence": str(evidence_path), "completedAt": datetime.now(timezone.utc).isoformat()})
527
+ job.setdefault("history", []).append({"event": "step-completed", "step": step, "evidence": str(evidence_path), "at": datetime.now(timezone.utc).isoformat()})
528
+ job["phase"] = "done" if all(item.get("status") == "done" for item in steps) else "ready"
529
+ save_design_job(project, job)
530
+ print(json.dumps({"jobId": job_id, "step": step, "status": target["status"], "phase": job["phase"], "evidence": str(evidence_path)}, indent=2))
531
+ return 0
532
+
533
+
534
+ def in_place_job(project: Path, routes: list[str], force: bool = False, content_source: str = "") -> int:
510
535
  """Create a resumable redesign contract for existing first-party routes."""
511
536
  project = project.resolve()
512
537
  if not routes:
@@ -540,7 +565,13 @@ def in_place_job(project: Path, routes: list[str], force: bool = False) -> int:
540
565
  break
541
566
  if not match:
542
567
  raise ValueError(f"existing route required for in-place redesign: {route}")
543
- resolved_routes.append({"route": value, "source": str(match)})
568
+ catch_all = match.stem in {"[...slug]", "[[...slug]]"}
569
+ resolved_routes.append({
570
+ "route": value,
571
+ "source": str(match),
572
+ "resolution": "declared" if content_source else ("unproven" if catch_all else "direct-file"),
573
+ "contentSource": content_source or ("unproven catch-all route; declare the DB/content resolver before implementation" if catch_all else str(match)),
574
+ })
544
575
  digest = hashlib.sha256((str(project) + "\n" + "\n".join(routes)).encode()).hexdigest()[:12]
545
576
  job_id = f"in-place-{digest}"
546
577
  path = design_job_path(project, job_id)
@@ -558,7 +589,34 @@ def in_place_job(project: Path, routes: list[str], force: bool = False) -> int:
558
589
  "clone_run_required": False,
559
590
  "shared_tokens_required": True,
560
591
  "palette_change_allowed": False,
561
- "required_steps": ["inspect-route", "apply-design-contract", "visual-review", "route-validation"],
592
+ "required_steps": ["inspect-route", "declare-content-source", "apply-design-contract", "visual-review", "route-validation"],
593
+ "steps": _step_records(["inspect-route", "declare-content-source", "apply-design-contract", "visual-review", "route-validation"]),
594
+ "history": [{"phase": "ready", "at": datetime.now(timezone.utc).isoformat()}],
595
+ }
596
+ save_design_job(project, job)
597
+ print(json.dumps(job, indent=2, ensure_ascii=False))
598
+ return 0
599
+
600
+
601
+ def section_job(project: Path, route: str, section_id: str, force: bool = False) -> int:
602
+ """Create a focused design job with an unchanged-rest assertion."""
603
+ project = project.resolve()
604
+ if not route.startswith("/") or not section_id.strip():
605
+ raise ValueError("section mode requires a route starting with / and a section id")
606
+ contract = require_design_contract(project)
607
+ digest = hashlib.sha256((str(project) + "\n" + route + "\n" + section_id).encode()).hexdigest()[:12]
608
+ job_id = f"section-{digest}"
609
+ path = design_job_path(project, job_id)
610
+ if path.exists() and not force:
611
+ print(path.read_text(encoding="utf-8"), end="")
612
+ return 0
613
+ required = ["inspect-section", "capture-before", "implement-section", "capture-after", "assert-unchanged-rest", "route-validation"]
614
+ job = {
615
+ "id": job_id, "workflow": "maggie-design", "mode": "section-scoped", "phase": "ready",
616
+ "project": str(project), "route": route, "sectionId": section_id,
617
+ "design_contract": contract, "clone_run_required": False,
618
+ "scope": {"targetSection": section_id, "unchangedRestAssertion": "all non-target section content and structure hashes remain unchanged"},
619
+ "required_steps": required, "steps": _step_records(required),
562
620
  "history": [{"phase": "ready", "at": datetime.now(timezone.utc).isoformat()}],
563
621
  }
564
622
  save_design_job(project, job)
@@ -566,6 +624,25 @@ def in_place_job(project: Path, routes: list[str], force: bool = False) -> int:
566
624
  return 0
567
625
 
568
626
 
627
+ def validate_section_evidence(before_path: Path, after_path: Path, section_id: str) -> dict[str, object]:
628
+ """Compare section-scoped evidence and prove the rest stayed unchanged."""
629
+ before = json.loads(before_path.resolve().read_text(encoding="utf-8"))
630
+ after = json.loads(after_path.resolve().read_text(encoding="utf-8"))
631
+ before_sections = {str(item.get("id")): item for item in before.get("sections", []) if isinstance(item, dict) and item.get("id")}
632
+ after_sections = {str(item.get("id")): item for item in after.get("sections", []) if isinstance(item, dict) and item.get("id")}
633
+ errors = []
634
+ if section_id not in before_sections or section_id not in after_sections:
635
+ errors.append("target section is missing from before or after evidence")
636
+ unchanged = sorted(section for section in before_sections.keys() & after_sections.keys() if section != section_id and before_sections[section] != after_sections[section])
637
+ if unchanged:
638
+ errors.append("non-target sections changed: " + ", ".join(unchanged))
639
+ missing_rest = sorted((before_sections.keys() ^ after_sections.keys()) - {section_id})
640
+ if missing_rest:
641
+ errors.append("non-target section set changed: " + ", ".join(missing_rest))
642
+ target_changed = section_id in before_sections and section_id in after_sections and before_sections[section_id] != after_sections[section_id]
643
+ return {"schemaVersion": "maggie-section-evidence.v1", "passed": not errors and target_changed, "targetSection": section_id, "targetChanged": target_changed, "unchangedRest": not unchanged and not missing_rest, "changedNonTargetSections": unchanged, "errors": errors + ([] if target_changed else ["target section did not change"]), "before": str(before_path.resolve()), "after": str(after_path.resolve())}
644
+
645
+
569
646
  def author_job(project: Path, route: str, purpose: str, audience: str, brief_file: Path | None, confirm: bool) -> int:
570
647
  """Create an original page brief without requiring an external source URL."""
571
648
  project = project.resolve()
@@ -592,6 +669,20 @@ def author_job(project: Path, route: str, purpose: str, audience: str, brief_fil
592
669
 
593
670
 
594
671
  def main() -> int:
672
+ if len(sys.argv) > 1 and sys.argv[1] in {"-h", "--help"}:
673
+ print("""usage: maggie_design.py <command> [options]
674
+
675
+ commands:
676
+ run run the complete clone-backed design workflow
677
+ in-place create a resumable redesign contract for existing routes
678
+ section create a section-scoped redesign contract
679
+ step record completion evidence for one design step
680
+ status show a design job and its per-step progress
681
+ resume restart a failed design workflow
682
+ author plan an original first-party page
683
+ reference-ui, init, validate-ui, rebrand, review
684
+ """)
685
+ return 0
595
686
  if len(sys.argv) > 1 and sys.argv[1] == "reference-ui":
596
687
  command = argparse.ArgumentParser(description="Record sanitized blog/service UI evidence from a read-only reference project.")
597
688
  command.add_argument("--project", type=Path, default=Path.cwd())
@@ -653,13 +744,51 @@ def main() -> int:
653
744
  in_place = argparse.ArgumentParser(description="Create an in-place redesign contract for existing first-party routes.")
654
745
  in_place.add_argument("--project", type=Path, default=Path.cwd())
655
746
  in_place.add_argument("--route", action="append", required=True, help="existing route, for example /pricing")
747
+ in_place.add_argument("--content-source", default="", help="declared DB/content resolver or source manifest for catch-all routes")
656
748
  in_place.add_argument("--force", action="store_true")
657
749
  args = in_place.parse_args(sys.argv[2:])
658
750
  try:
659
- return in_place_job(args.project, args.route, args.force)
751
+ return in_place_job(args.project, args.route, args.force, args.content_source)
660
752
  except (OSError, ValueError, json.JSONDecodeError) as error:
661
753
  print(f"BLOCKED: maggie-design in-place: {error}", file=sys.stderr)
662
754
  return 1
755
+ if len(sys.argv) > 1 and sys.argv[1] == "section":
756
+ section = argparse.ArgumentParser(description="Create a section-scoped redesign contract with an unchanged-rest assertion.")
757
+ section.add_argument("--project", type=Path, default=Path.cwd())
758
+ section.add_argument("--route", required=True)
759
+ section.add_argument("--section-id", required=True)
760
+ section.add_argument("--force", action="store_true")
761
+ args = section.parse_args(sys.argv[2:])
762
+ try:
763
+ return section_job(args.project, args.route, args.section_id, args.force)
764
+ except (OSError, ValueError, json.JSONDecodeError) as error:
765
+ print(f"BLOCKED: maggie-design section: {error}", file=sys.stderr)
766
+ return 1
767
+ if len(sys.argv) > 1 and sys.argv[1] == "step":
768
+ step = argparse.ArgumentParser(description="Record evidence for one design workflow step.")
769
+ step.add_argument("job_id")
770
+ step.add_argument("--project", type=Path, default=Path.cwd())
771
+ step.add_argument("--step", required=True)
772
+ step.add_argument("--evidence", type=Path, required=True)
773
+ args = step.parse_args(sys.argv[2:])
774
+ try:
775
+ return design_step(args.project, args.job_id, args.step, str(args.evidence))
776
+ except (OSError, ValueError, json.JSONDecodeError) as error:
777
+ print(f"BLOCKED: maggie-design step: {error}", file=sys.stderr)
778
+ return 1
779
+ if len(sys.argv) > 1 and sys.argv[1] == "section-validate":
780
+ section_validate = argparse.ArgumentParser(description="Validate before/after evidence for one section and assert the rest is unchanged.")
781
+ section_validate.add_argument("--before", type=Path, required=True)
782
+ section_validate.add_argument("--after", type=Path, required=True)
783
+ section_validate.add_argument("--section-id", required=True)
784
+ args = section_validate.parse_args(sys.argv[2:])
785
+ try:
786
+ result = validate_section_evidence(args.before, args.after, args.section_id)
787
+ print(json.dumps(result, indent=2, ensure_ascii=False))
788
+ return 0 if result["passed"] else 1
789
+ except (OSError, ValueError, json.JSONDecodeError) as error:
790
+ print(f"BLOCKED: maggie-design section-validate: {error}", file=sys.stderr)
791
+ return 1
663
792
  if len(sys.argv) > 1 and sys.argv[1] == "author":
664
793
  author = argparse.ArgumentParser(description="Create an original first-party page brief and route plan.")
665
794
  author.add_argument("--project", type=Path, default=Path.cwd())
@@ -204,6 +204,9 @@ def main() -> int:
204
204
  parser.add_argument("--project", default=".")
205
205
  sub = parser.add_subparsers(dest="command", required=True)
206
206
  collect_parser = sub.add_parser("collect")
207
+ # Accept the project flag after the subcommand as shown in the public
208
+ # examples. argparse otherwise only accepts the global flag before it.
209
+ collect_parser.add_argument("--project", default=argparse.SUPPRESS)
207
210
  collect_parser.add_argument("--skill", default="")
208
211
  collect_parser.add_argument("--run-id", default="")
209
212
  collect_parser.add_argument("--run-report")
@@ -10,11 +10,19 @@ from pathlib import Path
10
10
 
11
11
  SOURCE_EXTENSIONS = {".astro", ".css", ".html", ".jsx", ".js", ".json", ".svelte", ".tsx", ".ts", ".vue"}
12
12
  IGNORE_NAMES = {"icon", "icons", "true", "false", "null", "undefined"}
13
+ NOISE_NAME = re.compile(r"^(?:inline-)?svg[-_]\d+$|^(?:solid|regular|brands|light|thin|duotone)[-_]\d+$|^\d+$", re.I)
14
+
15
+
16
+ def normalize_name(name: str) -> str:
17
+ """Return the family-neutral name used for source/runtime comparison."""
18
+ value = name.strip().lower().replace("/", ":")
19
+ value = re.sub(r"^(?:phosphor|ph|font-awesome|fa|lucide):", "", value)
20
+ return value
13
21
 
14
22
 
15
23
  def _add(found: dict[str, set[str]], name: str, file: Path, evidence: str) -> None:
16
- name = name.strip().lower()
17
- if not name or name in IGNORE_NAMES or len(name) > 80 or not re.fullmatch(r"[a-z0-9][a-z0-9._:/-]*", name):
24
+ name = normalize_name(name)
25
+ if not name or name in IGNORE_NAMES or NOISE_NAME.fullmatch(name) or len(name) > 80 or not re.fullmatch(r"[a-z0-9][a-z0-9._:-]*", name):
18
26
  return
19
27
  found.setdefault(name, set()).add(f"{file.as_posix()}:{evidence}")
20
28
 
@@ -31,7 +39,7 @@ def source_icons(root: Path, source_dir: Path) -> dict[str, set[str]]:
31
39
  _add(found, match.group(1), relative, "named-property")
32
40
  for match in re.finditer(r"data-(?:icon|glyph)\s*=\s*[\"']([^\"']+)", text, re.I):
33
41
  _add(found, match.group(1), relative, "data-attribute")
34
- for match in re.finditer(r"(?:ph|fa|lucide|icon)-([a-z0-9][a-z0-9-]*)", text, re.I):
42
+ for match in re.finditer(r"(?:ph|fa|lucide)-([a-z0-9][a-z0-9-]*)", text, re.I):
35
43
  _add(found, match.group(1), relative, "icon-class")
36
44
  for match in re.finditer(r"(?:phosphor|lucide|icon)[:/]([a-z0-9][a-z0-9-]*)", text, re.I):
37
45
  _add(found, match.group(1), relative, "icon-token")
@@ -49,10 +57,14 @@ def runtime_icons(root: Path, runtime_paths: list[Path]) -> tuple[set[str], dict
49
57
  continue
50
58
  text = path.read_text(encoding="utf-8", errors="replace")
51
59
  relative = path.relative_to(root.resolve()) if path.is_relative_to(root.resolve()) else path
52
- for match in re.finditer(r"\.(?:ph|fa|lucide|icon)-([a-z0-9][a-z0-9-]*)\b", text, re.I):
53
- name = match.group(1).lower(); names.add(name); evidence.setdefault(name, set()).add(f"{relative}:css-class")
60
+ for match in re.finditer(r"\.(?:ph|fa|lucide)-([a-z0-9][a-z0-9-]*)\b", text, re.I):
61
+ name = normalize_name(match.group(1))
62
+ if not NOISE_NAME.fullmatch(name):
63
+ names.add(name); evidence.setdefault(name, set()).add(f"{relative}:css-class")
54
64
  for match in re.finditer(r"(?:data-(?:icon|glyph)|icon)\s*=\s*[\"']([^\"']+)", text, re.I):
55
- name = match.group(1).lower(); names.add(name); evidence.setdefault(name, set()).add(f"{relative}:runtime-map")
65
+ name = normalize_name(match.group(1))
66
+ if name not in IGNORE_NAMES and not NOISE_NAME.fullmatch(name):
67
+ names.add(name); evidence.setdefault(name, set()).add(f"{relative}:runtime-map")
56
68
  return names, evidence, font_files
57
69
 
58
70
 
@@ -127,6 +127,60 @@ def evidence_gate(project: Path, environment: str) -> dict:
127
127
  return {"name": "durable-site-evidence", "passed": not errors, "exitCode": 0 if not errors else 1, "result": {"passed": not errors, "errors": errors, "evidence": summaries}, "stderr": ""}
128
128
 
129
129
 
130
+ def changed_surface_gate(project: Path) -> dict:
131
+ """Require visual/runtime evidence when a visitor-facing surface changed."""
132
+ try:
133
+ changed = subprocess.run(
134
+ ["git", "-C", str(project), "diff", "--name-only"],
135
+ capture_output=True, text=True, check=True,
136
+ ).stdout.splitlines()
137
+ except (OSError, subprocess.CalledProcessError):
138
+ return {"name": "changed-surface-evidence", "passed": True, "exitCode": 0,
139
+ "result": {"passed": True, "changed": False, "reason": "project is not a git worktree"}, "stderr": ""}
140
+ visitor_suffixes = {".astro", ".css", ".scss", ".html", ".jsx", ".tsx", ".js", ".ts", ".svg", ".png", ".jpg", ".jpeg", ".webp"}
141
+ surfaces = sorted(path for path in changed if Path(path).suffix.lower() in visitor_suffixes)
142
+ if not surfaces:
143
+ return {"name": "changed-surface-evidence", "passed": True, "exitCode": 0,
144
+ "result": {"passed": True, "changed": False, "surfaces": []}, "stderr": ""}
145
+
146
+ candidates = {
147
+ "iconInventory": (project / "docs/icon-inventory.json", project / ".maggie/icon-inventory.json"),
148
+ "renderEvidence": (project / "docs/deployment-canary.json", project / ".maggie/deployment-canary.json", project / "docs/rendered-canary.json", project / ".maggie/rendered-canary.json"),
149
+ }
150
+ errors: list[str] = []
151
+ evidence: dict[str, dict[str, object]] = {}
152
+ for name, paths in candidates.items():
153
+ path = next((candidate for candidate in paths if candidate.exists()), None)
154
+ if path is None:
155
+ errors.append(f"missing {name} evidence")
156
+ continue
157
+ try:
158
+ payload = json.loads(path.read_text(encoding="utf-8"))
159
+ except (OSError, json.JSONDecodeError) as exc:
160
+ errors.append(f"{name}: {exc}")
161
+ continue
162
+ passed = payload.get("passed") is True or payload.get("status") in {"pass", "passed"}
163
+ evidence[name] = {"path": str(path.relative_to(project)), "passed": passed}
164
+ if not passed:
165
+ errors.append(f"{name} is not passed")
166
+ if name == "renderEvidence":
167
+ render = payload.get("render", payload)
168
+ screenshots = render.get("screenshots", render.get("screenshotEvidence", [])) if isinstance(render, dict) else []
169
+ console_errors = render.get("consoleErrors", []) if isinstance(render, dict) else []
170
+ network_errors = render.get("networkErrors", []) if isinstance(render, dict) else []
171
+ placeholders = render.get("placeholderMatches", render.get("placeholders", [])) if isinstance(render, dict) else []
172
+ if not screenshots:
173
+ errors.append("renderEvidence has no screenshot evidence")
174
+ if console_errors:
175
+ errors.append("renderEvidence contains console errors")
176
+ if network_errors:
177
+ errors.append("renderEvidence contains network errors")
178
+ if placeholders:
179
+ errors.append("renderEvidence contains placeholder matches")
180
+ return {"name": "changed-surface-evidence", "passed": not errors, "exitCode": 0 if not errors else 1,
181
+ "result": {"passed": not errors, "changed": True, "surfaces": surfaces, "evidence": evidence, "errors": errors}, "stderr": ""}
182
+
183
+
130
184
  def editorial_gate(project: Path) -> dict:
131
185
  """Require explicit editorial approval for every launch category.
132
186
 
@@ -224,6 +278,7 @@ def main() -> int:
224
278
  parser.error(f"project directory does not exist: {project}")
225
279
 
226
280
  gates = []
281
+ gates.append(changed_surface_gate(project))
227
282
  gates.append(build_gate(project))
228
283
  for name, command in build_gates(project, args.environment, args.target, not args.skip_compatibility):
229
284
  gates.append(run_gate(name, command, project))
@@ -9,9 +9,8 @@ import json
9
9
  import re
10
10
  import sys
11
11
  from pathlib import Path
12
- from urllib.parse import urljoin
13
- from html.parser import HTMLParser
14
12
  from urllib.parse import urljoin, urlparse
13
+ from html.parser import HTMLParser
15
14
  from urllib.request import Request, urlopen
16
15
 
17
16
  sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "runtime"))
@@ -39,6 +38,10 @@ class PageParser(HTMLParser):
39
38
  self.visible_text = []
40
39
  self.hidden_depth = 0
41
40
  self.templates = set()
41
+ self._content_capture = []
42
+ self.headings = []
43
+ self.paragraphs = []
44
+ self.links = []
42
45
 
43
46
  def handle_starttag(self, tag, attrs):
44
47
  data = dict(attrs)
@@ -49,6 +52,8 @@ class PageParser(HTMLParser):
49
52
  if not self.hidden_depth:
50
53
  self.structure.append(["start", tag, {key: data[key] for key in
51
54
  ("class", "id", "role", "href", "src", "data-template", "data-section", "data-i18n") if key in data}])
55
+ if tag in {"h1", "h2", "h3", "h4", "h5", "h6", "p", "a"}:
56
+ self._content_capture.append({"tag": tag, "parts": [], "href": data.get("href", "")})
52
57
  if tag == "html":
53
58
  self.lang = data.get("lang", "")
54
59
  if tag == "meta" and data.get("name"):
@@ -74,6 +79,15 @@ class PageParser(HTMLParser):
74
79
  self.images.append({"src": data.get("src", ""), "alt": data.get("alt")})
75
80
 
76
81
  def handle_endtag(self, tag):
82
+ if not self.hidden_depth and self._content_capture and self._content_capture[-1]["tag"] == tag:
83
+ capture = self._content_capture.pop()
84
+ text = " ".join("".join(capture["parts"]).split())
85
+ if text and tag.startswith("h"):
86
+ self.headings.append(text)
87
+ elif text and tag == "p":
88
+ self.paragraphs.append(text)
89
+ elif tag == "a" and capture.get("href"):
90
+ self.links.append({"href": capture["href"], "text": text})
77
91
  if tag in {"script", "style", "noscript"}:
78
92
  self.hidden_depth = max(0, self.hidden_depth - 1)
79
93
  elif not self.hidden_depth:
@@ -91,6 +105,8 @@ class PageParser(HTMLParser):
91
105
  def handle_data(self, data):
92
106
  if not self.hidden_depth and data.strip():
93
107
  self.visible_text.append(" ".join(data.split()))
108
+ for capture in self._content_capture:
109
+ capture["parts"].append(data)
94
110
  if self.in_title:
95
111
  self.title += data.strip()
96
112
  if self._jsonld is not None:
@@ -148,6 +164,22 @@ def audit_page(url: str, html: str, status: int, content_type: str, expected_lan
148
164
  "structureHash": hashlib.sha256(json.dumps(page.structure, sort_keys=True).encode()).hexdigest(),
149
165
  "textHash": hashlib.sha256(" ".join(page.visible_text).encode()).hexdigest(),
150
166
  },
167
+ "contentContract": {
168
+ "headings": page.headings,
169
+ "paragraphs": page.paragraphs,
170
+ "links": [{"href": urljoin(url, link["href"]), "text": link["text"]} for link in page.links],
171
+ # Images and their alt text are readable content too. Keeping them
172
+ # in this contract prevents a prose-only migration from silently
173
+ # dropping the visual content of a page.
174
+ "images": [{"src": urljoin(url, image["src"]), "alt": image.get("alt")} for image in page.images],
175
+ "imageAlts": [image.get("alt") for image in page.images],
176
+ "contentHash": hashlib.sha256(json.dumps({
177
+ "headings": page.headings,
178
+ "paragraphs": page.paragraphs,
179
+ "links": [{"href": urljoin(url, link["href"]), "text": link["text"]} for link in page.links],
180
+ "images": [{"src": urljoin(url, image["src"]), "alt": image.get("alt")} for image in page.images],
181
+ }, sort_keys=True, ensure_ascii=False).encode()).hexdigest(),
182
+ },
151
183
  "robots": not robots or not any(token in {"noindex", "none", "nofollow"} for token in robots_tokens),
152
184
  "robots_directives": robots,
153
185
  "robots_conflict": len({token for token in robots_tokens if token in {"index", "noindex", "follow", "nofollow", "none"}} & {"index", "noindex"}) > 1 or len({token for token in robots_tokens if token in {"follow", "nofollow", "none"}} & {"follow", "nofollow"}) > 1,
@@ -12,6 +12,10 @@ from typing import Any, Iterable
12
12
 
13
13
  SCHEMA = "maggiedash-section-registry.v1"
14
14
  IDENTITY_SCHEMA = "maggiedash-section-identity.v1"
15
+ TRANSLATION_REMAP_SCHEMA = "maggiedash-translation-remap.v1"
16
+ FANOUT_SCHEMA = "maggiedash-section-fanout.v1"
17
+ RECONCILE_SCHEMA = "maggiedash-reconcile.v1"
18
+ LOCALE_SCHEMA = "maggiedash-locale-coverage.v1"
15
19
 
16
20
 
17
21
  def _validate_field(field: object, at: str, errors: list[str], *, nested: bool = False) -> None:
@@ -153,6 +157,72 @@ def copy_notes(registry: dict[str, Any], sections: object) -> dict[str, Any]:
153
157
  return {"passed": not errors, "errors": errors, "notes": notes}
154
158
 
155
159
 
160
+ def validate_values(registry: dict[str, Any], sections: object) -> dict[str, Any]:
161
+ """Validate required scalar and repeated values, including pairs rows."""
162
+ if not isinstance(sections, list):
163
+ return {"passed": False, "errors": ["sections must be a list"]}
164
+ specs = {str(item.get("type")): item for item in registry.get("sections", []) if isinstance(item, dict)}
165
+ errors: list[str] = []
166
+
167
+ def check_fields(value: object, fields: list[dict[str, Any]], path: str) -> None:
168
+ for field in fields:
169
+ name = str(field.get("name"))
170
+ child = value.get(name) if isinstance(value, dict) else None
171
+ at = f"{path}.{name}"
172
+ if field.get("required") and (child is None or (isinstance(child, str) and not child.strip())):
173
+ errors.append(f"{at} is required and cannot be blank")
174
+ repeats = field.get("repeats")
175
+ if not isinstance(repeats, dict) or not isinstance(child, list):
176
+ continue
177
+ if not repeats["min"] <= len(child) <= repeats["max"]:
178
+ errors.append(f"{at} must contain {repeats['min']}-{repeats['max']} entries")
179
+ child_fields = [item for item in repeats.get("of", []) if isinstance(item, dict)]
180
+ for index, item in enumerate(child):
181
+ if isinstance(item, dict):
182
+ check_fields(item, child_fields, f"{at}[{index}]")
183
+ elif len(child_fields) == 1 and child_fields[0].get("required") and not str(item).strip():
184
+ errors.append(f"{at}[{index}] is required and cannot be blank")
185
+
186
+ for index, section in enumerate(sections):
187
+ if not isinstance(section, dict):
188
+ errors.append(f"sections[{index}] must be an object")
189
+ continue
190
+ spec = specs.get(str(section.get("type")))
191
+ if spec:
192
+ check_fields(section, [field for field in spec.get("fields", []) if isinstance(field, dict)], f"sections[{index}]")
193
+ return {"passed": not errors, "errors": errors}
194
+
195
+
196
+ def translation_key_paths(page_id: str, sections: object, field_names: tuple[str, ...] = ("imageAlt",)) -> list[str]:
197
+ """Find stored section fields that need a locale sidecar value."""
198
+ if not isinstance(sections, list):
199
+ return []
200
+ paths: list[str] = []
201
+
202
+ def walk(value: object, path: str) -> None:
203
+ if isinstance(value, dict):
204
+ for key, child in value.items():
205
+ child_path = f"{path}.{key}"
206
+ if key in field_names and isinstance(child, str) and child.strip():
207
+ paths.append(child_path)
208
+ walk(child, child_path)
209
+ elif isinstance(value, list):
210
+ for index, child in enumerate(value):
211
+ walk(child, f"{path}.{index}")
212
+
213
+ for index, section in enumerate(sections):
214
+ walk(section, f"page.{page_id}.sections.{section_key(section, index)}")
215
+ return paths
216
+
217
+
218
+ def validate_locale_coverage(page_id: str, sections: object, translations: object, locales: Iterable[str]) -> dict[str, Any]:
219
+ """Require non-default locale rows for every declared translatable field."""
220
+ paths = translation_key_paths(page_id, sections)
221
+ values = translations if isinstance(translations, dict) else {}
222
+ missing = [{"locale": locale, "key": key} for locale in locales for key in paths if not isinstance(values.get(locale), dict) or not str(values[locale].get(key) or "").strip()]
223
+ return {"schemaVersion": LOCALE_SCHEMA, "passed": not missing, "pageId": page_id, "requiredKeys": paths, "missing": missing, "writePolicy": "section row and locale rows must be committed in one transaction"}
224
+
225
+
156
226
  def new_section_id(existing: Iterable[str] = ()) -> str:
157
227
  taken = set(existing)
158
228
  while True:
@@ -191,3 +261,73 @@ def section_id_migration(page_id: str, sections: list[dict[str, Any]]) -> dict[s
191
261
  identified = ensure_section_ids(sections)
192
262
  mappings = [{"from": f"page.{page_id}.sections.{index}.", "to": f"page.{page_id}.sections.{section['id']}."} for index, section in enumerate(identified)]
193
263
  return {"schemaVersion": IDENTITY_SCHEMA, "pageId": page_id, "sections": identified, "renameOrder": sorted(mappings, key=lambda item: len(item["from"]), reverse=True), "sourceChecksum": hashlib.sha256(json.dumps(sections, sort_keys=True, ensure_ascii=False, separators=(",", ":")).encode()).hexdigest()}
264
+
265
+
266
+ def remap_translations(translations: object, mappings: object) -> dict[str, Any]:
267
+ """Move an existing translation keyspace without dropping unmapped keys."""
268
+ if not isinstance(translations, dict):
269
+ return {"schemaVersion": TRANSLATION_REMAP_SCHEMA, "passed": False, "errors": ["translations must be an object"]}
270
+ if isinstance(mappings, dict):
271
+ mappings = mappings.get("renameOrder") or mappings.get("mappings")
272
+ if not isinstance(mappings, list) or not mappings:
273
+ return {"schemaVersion": TRANSLATION_REMAP_SCHEMA, "passed": False, "errors": ["mappings must be a non-empty list"]}
274
+ valid = [item for item in mappings if isinstance(item, dict) and isinstance(item.get("from"), str) and isinstance(item.get("to"), str) and item["from"] and item["to"]]
275
+ if len(valid) != len(mappings):
276
+ return {"schemaVersion": TRANSLATION_REMAP_SCHEMA, "passed": False, "errors": ["every mapping needs non-empty from and to prefixes"]}
277
+ ordered = sorted(valid, key=lambda item: len(item["from"]), reverse=True)
278
+ result: dict[str, Any] = {}
279
+ applied: list[dict[str, str]] = []
280
+ unmapped: list[str] = []
281
+ collisions: list[str] = []
282
+ for key, value in translations.items():
283
+ source = str(key)
284
+ destination = next((item["to"] + source[len(item["from"]):] for item in ordered if source.startswith(item["from"])), None)
285
+ if destination is None:
286
+ destination = source
287
+ unmapped.append(source)
288
+ elif destination != source:
289
+ applied.append({"from": source, "to": destination})
290
+ if destination in result and result[destination] != value:
291
+ collisions.append(destination)
292
+ continue
293
+ result[destination] = value
294
+ errors = [f"translation key collision: {key}" for key in sorted(set(collisions))]
295
+ return {"schemaVersion": TRANSLATION_REMAP_SCHEMA, "passed": not errors, "errors": errors,
296
+ "translations": result, "applied": applied, "unmapped": sorted(unmapped),
297
+ "mappingCount": len(valid)}
298
+
299
+
300
+ def validate_fanout(registry: dict[str, Any], fanout: object) -> dict[str, Any]:
301
+ """Ensure every registry type is wired through each host implementation surface.
302
+
303
+ A registry is declarative, but hosts still need a type union/schema,
304
+ blank-state logic, validation, text extraction, renderer, and editor
305
+ affordance. This contract makes a missing fan-out edit fail loudly.
306
+ """
307
+ if not isinstance(fanout, dict) or fanout.get("schemaVersion") != FANOUT_SCHEMA:
308
+ return {"schemaVersion": FANOUT_SCHEMA, "passed": False, "errors": [f"fanout schemaVersion must be {FANOUT_SCHEMA}"]}
309
+ types = {str(item.get("type")) for item in registry.get("sections", []) if isinstance(item, dict)}
310
+ surfaces = ("typeUnion", "sectionSchema", "blank", "validation", "textExtraction", "renderer", "editor")
311
+ missing: list[dict[str, str]] = []
312
+ for surface in surfaces:
313
+ values = fanout.get(surface)
314
+ declared = set(values) if isinstance(values, list) else set(values.keys()) if isinstance(values, dict) else set()
315
+ for section_type in sorted(types - declared):
316
+ missing.append({"type": section_type, "surface": surface})
317
+ return {"schemaVersion": FANOUT_SCHEMA, "passed": not missing, "errors": [f"{item['type']} missing {item['surface']} fan-out" for item in missing], "sectionTypes": sorted(types), "missing": missing}
318
+
319
+
320
+ def reconcile_fields(current: object, desired: object) -> dict[str, Any]:
321
+ """Return an idempotent compare-and-set result for repair scripts."""
322
+ if not isinstance(current, dict) or not isinstance(desired, dict):
323
+ return {"schemaVersion": RECONCILE_SCHEMA, "passed": False, "errors": ["current and desired must be objects"]}
324
+ result = copy.deepcopy(current)
325
+ changed: list[str] = []
326
+ already_correct: list[str] = []
327
+ for key, value in desired.items():
328
+ if current.get(key) == value:
329
+ already_correct.append(str(key))
330
+ else:
331
+ result[key] = copy.deepcopy(value)
332
+ changed.append(str(key))
333
+ return {"schemaVersion": RECONCILE_SCHEMA, "passed": True, "changed": changed, "alreadyCorrect": already_correct, "result": result, "convergent": True}
@@ -5,6 +5,7 @@ from __future__ import annotations
5
5
  import hashlib
6
6
  import re
7
7
  from pathlib import Path
8
+ from typing import Any
8
9
 
9
10
 
10
11
  IMPORT_RE = re.compile(r"(?:import\s+(?:[^;\n]*?\s+from\s+)?|export\s+[^;\n]*?\s+from\s+)[\"']([^\"']+)[\"']")
@@ -49,3 +50,25 @@ def imported_components(route: Path) -> list[dict[str, str | None]]:
49
50
  "fingerprint": hashlib.sha256(resolved.read_bytes()).hexdigest() if resolved else None,
50
51
  })
51
52
  return result
53
+
54
+
55
+ def classify_bindings(bindings: list[dict[str, Any]], section_paths: set[str], page_paths: set[str]) -> dict[str, Any]:
56
+ """Classify component bindings against section and page route inventories.
57
+
58
+ Hosts supply the inventories from their renderer/database. This keeps the
59
+ report honest: a binding is not called "missing" merely because a legacy
60
+ component file still exists on disk.
61
+ """
62
+ records = []
63
+ for binding in bindings:
64
+ route = str(binding.get("route") or binding.get("path") or "")
65
+ component = str(binding.get("component") or "")
66
+ if route in section_paths or component in section_paths or binding.get("superseded") is True:
67
+ state = "superseded"
68
+ elif route in page_paths or binding.get("pageId"):
69
+ state = "migratable"
70
+ else:
71
+ state = "no-page-row"
72
+ records.append({**binding, "state": state})
73
+ counts = {state: sum(item["state"] == state for item in records) for state in ("superseded", "migratable", "no-page-row")}
74
+ return {"schemaVersion": "maggie-component-binding-audit.v1", "passed": not any(item["state"] == "no-page-row" for item in records), "counts": counts, "bindings": records}