@topy-ai/maggie 0.7.12 → 0.7.14
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +43 -11
- package/README.zh-TW.md +19 -3
- package/bin/maggie.js +5 -1
- package/bundled-contracts/maggiedash/README.md +3 -0
- package/bundled-contracts/maggiedash/quality-contracts.md +58 -0
- package/bundled-skills/README.md +1 -0
- package/bundled-skills/catalog.json +4 -0
- package/bundled-skills/maggie-dash/SKILL.md +31 -8
- package/bundled-skills/maggie-deployment/SKILL.md +15 -0
- package/bundled-skills/maggie-qa-workflow/SKILL.md +102 -0
- package/bundled-skills/maggie-seo-geo/SKILL.md +1 -1
- package/bundled-templates/maggiedash/section-fanout.json +84 -8
- package/bundled-templates/maggiedash/section-registry.json +3 -3
- package/bundled-tools/clis/maggie_dash.py +48 -2
- package/bundled-tools/clis/maggie_migration.py +67 -0
- package/bundled-tools/clis/maggie_qa_workflow.py +367 -0
- package/bundled-tools/runtime/maggie_quality.py +231 -0
- package/bundled-tools/runtime/maggie_sections.py +112 -12
- package/bundled-tools/runtime/maggie_sitemap.py +9 -1
- package/package.json +1 -1
|
@@ -0,0 +1,231 @@
|
|
|
1
|
+
"""Provider-neutral quality contracts for variants, media, inventory, and bindings."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import hashlib
|
|
6
|
+
import json
|
|
7
|
+
import re
|
|
8
|
+
from collections import Counter, defaultdict
|
|
9
|
+
from typing import Any, Iterable
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
VARIANT_SCHEMA = "maggie-variant-copy.v1"
|
|
13
|
+
MEDIA_SCHEMA = "maggie-media-uniqueness.v1"
|
|
14
|
+
INVENTORY_SCHEMA = "maggie-page-inventory.v1"
|
|
15
|
+
BINDING_SCHEMA = "maggie-section-bindings.v1"
|
|
16
|
+
IDEMPOTENCY_SCHEMA = "maggie-reconcile-contract.v1"
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def _items(value: object, *keys: str) -> list[dict[str, Any]]:
|
|
20
|
+
if isinstance(value, list):
|
|
21
|
+
return [item for item in value if isinstance(item, dict)]
|
|
22
|
+
if isinstance(value, dict):
|
|
23
|
+
for key in keys:
|
|
24
|
+
candidate = value.get(key)
|
|
25
|
+
if isinstance(candidate, list):
|
|
26
|
+
return [item for item in candidate if isinstance(item, dict)]
|
|
27
|
+
return []
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def _normalise(value: object) -> object:
|
|
31
|
+
if isinstance(value, dict):
|
|
32
|
+
return {str(key): _normalise(value[key]) for key in sorted(value)}
|
|
33
|
+
if isinstance(value, list):
|
|
34
|
+
values = [_normalise(item) for item in value]
|
|
35
|
+
return sorted(values, key=lambda item: json.dumps(item, sort_keys=True, ensure_ascii=False))
|
|
36
|
+
if isinstance(value, str):
|
|
37
|
+
return re.sub(r"\s+", " ", value).strip().casefold()
|
|
38
|
+
return value
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _field_value(page: dict[str, Any], field: str) -> object:
|
|
42
|
+
if field in page:
|
|
43
|
+
return page[field]
|
|
44
|
+
for section in page.get("sections", []) if isinstance(page.get("sections"), list) else []:
|
|
45
|
+
if isinstance(section, dict) and (section.get("type") == field or section.get("id") == field):
|
|
46
|
+
copy = {key: value for key, value in section.items() if key not in {"id", "type"}}
|
|
47
|
+
return copy
|
|
48
|
+
return None
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def validate_variant_copy(value: object, fields: Iterable[str] = ("faq",)) -> dict[str, Any]:
|
|
52
|
+
pages = _items(value, "pages", "variants")
|
|
53
|
+
errors: list[str] = []
|
|
54
|
+
duplicate_groups: list[dict[str, Any]] = []
|
|
55
|
+
grouped: dict[tuple[str, str, str], list[dict[str, Any]]] = defaultdict(list)
|
|
56
|
+
field_list = [str(field) for field in fields if str(field)]
|
|
57
|
+
for index, page in enumerate(pages):
|
|
58
|
+
page_id = str(page.get("id") or page.get("variantId") or f"item-{index}")
|
|
59
|
+
family = str(page.get("family") or page.get("variantFamily") or page.get("sourceVariantKey") or page.get("canonicalVariantId") or "")
|
|
60
|
+
locale = str(page.get("locale") or "default")
|
|
61
|
+
if not family:
|
|
62
|
+
errors.append(f"{page_id}: variant family is required")
|
|
63
|
+
continue
|
|
64
|
+
for field in field_list:
|
|
65
|
+
selected = _field_value(page, field)
|
|
66
|
+
if selected is not None:
|
|
67
|
+
digest = hashlib.sha256(json.dumps(_normalise(selected), sort_keys=True, ensure_ascii=False).encode()).hexdigest()
|
|
68
|
+
grouped[(family, locale, field)].append({"id": page_id, "digest": digest})
|
|
69
|
+
for (family, locale, field), records in sorted(grouped.items()):
|
|
70
|
+
by_digest: dict[str, list[str]] = defaultdict(list)
|
|
71
|
+
for record in records:
|
|
72
|
+
by_digest[record["digest"]].append(record["id"])
|
|
73
|
+
for digest, ids in sorted(by_digest.items()):
|
|
74
|
+
if len(ids) > 1:
|
|
75
|
+
duplicate_groups.append({"family": family, "locale": locale, "field": field, "variantIds": sorted(ids), "digest": digest})
|
|
76
|
+
return {"schemaVersion": VARIANT_SCHEMA, "passed": not errors and not duplicate_groups, "errors": errors, "duplicates": duplicate_groups, "pages": len(pages), "fields": field_list}
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def _media_sources(value: object, path: str = "") -> list[dict[str, str]]:
|
|
80
|
+
media_keys = {"image", "imageurl", "image_url", "src", "srcset", "poster", "media", "mediaurl", "media_url"}
|
|
81
|
+
result: list[dict[str, str]] = []
|
|
82
|
+
if isinstance(value, dict):
|
|
83
|
+
for key, child in value.items():
|
|
84
|
+
child_path = f"{path}.{key}" if path else str(key)
|
|
85
|
+
if str(key).casefold() in media_keys and isinstance(child, str) and child.strip():
|
|
86
|
+
result.append({"source": child.strip(), "path": child_path})
|
|
87
|
+
else:
|
|
88
|
+
result.extend(_media_sources(child, child_path))
|
|
89
|
+
elif isinstance(value, list):
|
|
90
|
+
for index, child in enumerate(value):
|
|
91
|
+
result.extend(_media_sources(child, f"{path}.{index}"))
|
|
92
|
+
return result
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def validate_media_uniqueness(value: object, *, across_siblings: bool = False) -> dict[str, Any]:
|
|
96
|
+
pages = _items(value, "pages", "variants")
|
|
97
|
+
errors: list[dict[str, Any]] = []
|
|
98
|
+
cross: dict[tuple[str, str], list[str]] = defaultdict(list)
|
|
99
|
+
for index, page in enumerate(pages):
|
|
100
|
+
page_id = str(page.get("id") or page.get("variantId") or f"item-{index}")
|
|
101
|
+
sources = _media_sources(page.get("sections", page))
|
|
102
|
+
by_source: dict[str, list[str]] = defaultdict(list)
|
|
103
|
+
for item in sources:
|
|
104
|
+
by_source[item["source"]].append(item["path"])
|
|
105
|
+
for source, paths in sorted(by_source.items()):
|
|
106
|
+
if len(paths) > 1:
|
|
107
|
+
errors.append({"scope": "page", "pageId": page_id, "source": source, "paths": paths})
|
|
108
|
+
if across_siblings:
|
|
109
|
+
family = str(page.get("family") or page.get("variantFamily") or page.get("sourceVariantKey") or page_id)
|
|
110
|
+
for source in by_source:
|
|
111
|
+
cross[(family, source)].append(page_id)
|
|
112
|
+
if across_siblings:
|
|
113
|
+
for (family, source), ids in sorted(cross.items()):
|
|
114
|
+
if len(set(ids)) > 1:
|
|
115
|
+
errors.append({"scope": "siblings", "family": family, "source": source, "pageIds": sorted(set(ids))})
|
|
116
|
+
return {"schemaVersion": MEDIA_SCHEMA, "passed": not errors, "duplicates": errors, "pages": len(pages), "acrossSiblings": across_siblings}
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def classify_inventory(value: object) -> dict[str, Any]:
|
|
120
|
+
pages = _items(value, "pages", "items", "routes")
|
|
121
|
+
allowed = {"band", "code-rendered", "redirect", "asset", "unknown"}
|
|
122
|
+
records: list[dict[str, Any]] = []
|
|
123
|
+
errors: list[str] = []
|
|
124
|
+
seen: set[str] = set()
|
|
125
|
+
for index, page in enumerate(pages):
|
|
126
|
+
path = str(page.get("path") or page.get("url") or page.get("route") or f"item-{index}")
|
|
127
|
+
if path in seen:
|
|
128
|
+
errors.append(f"duplicate inventory path: {path}")
|
|
129
|
+
seen.add(path)
|
|
130
|
+
explicit = str(page.get("kind") or page.get("classification") or "").casefold()
|
|
131
|
+
if explicit in {"code", "code-rendered", "component", "source"}:
|
|
132
|
+
kind = "code-rendered"
|
|
133
|
+
elif explicit in {"band", "section", "sections"}:
|
|
134
|
+
kind = "band"
|
|
135
|
+
elif explicit in {"redirect", "asset"}:
|
|
136
|
+
kind = explicit
|
|
137
|
+
elif isinstance(page.get("sections"), list) or isinstance(page.get("sectionTypes"), list):
|
|
138
|
+
kind = "band"
|
|
139
|
+
elif page.get("renderedBy") or re.search(r"\.(astro|tsx|jsx|vue|svelte)$", str(page.get("source") or page.get("file") or path), re.I):
|
|
140
|
+
kind = "code-rendered"
|
|
141
|
+
else:
|
|
142
|
+
kind = "unknown"
|
|
143
|
+
records.append({"path": path, "kind": kind})
|
|
144
|
+
if kind not in allowed or kind == "unknown":
|
|
145
|
+
errors.append(f"{path}: cannot classify published page")
|
|
146
|
+
counts = dict(sorted(Counter(item["kind"] for item in records).items()))
|
|
147
|
+
return {"schemaVersion": INVENTORY_SCHEMA, "passed": not errors and len(records) == len(seen), "records": records, "counts": counts, "bandCount": counts.get("band", 0), "codeRenderedCount": counts.get("code-rendered", 0), "errors": errors, "disjoint": True}
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def _reference_values(references: object) -> dict[str, set[str]]:
|
|
151
|
+
source = references if isinstance(references, dict) else {}
|
|
152
|
+
result: dict[str, set[str]] = {}
|
|
153
|
+
for key, values in source.items():
|
|
154
|
+
entries = values if isinstance(values, list) else [values]
|
|
155
|
+
allowed: set[str] = set()
|
|
156
|
+
for item in entries:
|
|
157
|
+
if isinstance(item, dict):
|
|
158
|
+
for candidate in (item.get("id"), item.get("slug"), item.get("name"), item.get("title"), item.get("value")):
|
|
159
|
+
if candidate is not None and str(candidate).strip():
|
|
160
|
+
allowed.add(str(candidate).strip())
|
|
161
|
+
elif isinstance(item, (str, int)):
|
|
162
|
+
allowed.add(str(item).strip())
|
|
163
|
+
result[str(key).casefold()] = allowed
|
|
164
|
+
return result
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def validate_bindings(sections: object, references: object) -> dict[str, Any]:
|
|
168
|
+
allowed = _reference_values(references)
|
|
169
|
+
unresolved: list[dict[str, str]] = []
|
|
170
|
+
checked: list[dict[str, str]] = []
|
|
171
|
+
suffixes = ("id", "slug", "ref", "reference")
|
|
172
|
+
values = sections if isinstance(sections, list) else sections.get("sections", []) if isinstance(sections, dict) else []
|
|
173
|
+
for index, section in enumerate(values):
|
|
174
|
+
if not isinstance(section, dict):
|
|
175
|
+
continue
|
|
176
|
+
section_id = str(section.get("id") or index)
|
|
177
|
+
for key, value in section.items():
|
|
178
|
+
if not isinstance(value, (str, int)) or not str(value).strip():
|
|
179
|
+
continue
|
|
180
|
+
key_lower = str(key).casefold()
|
|
181
|
+
base = key_lower
|
|
182
|
+
for suffix in suffixes:
|
|
183
|
+
if base.endswith(suffix) and len(base) > len(suffix):
|
|
184
|
+
base = base[: -len(suffix)]
|
|
185
|
+
break
|
|
186
|
+
if base not in allowed:
|
|
187
|
+
continue
|
|
188
|
+
record = {"sectionId": section_id, "field": str(key), "value": str(value)}
|
|
189
|
+
checked.append(record)
|
|
190
|
+
if str(value) not in allowed[base]:
|
|
191
|
+
unresolved.append(record)
|
|
192
|
+
binding = section.get("binding")
|
|
193
|
+
if isinstance(binding, dict):
|
|
194
|
+
binding_type = str(binding.get("type") or "").casefold()
|
|
195
|
+
binding_value = binding.get("value", binding.get("slug", binding.get("id", binding.get("ref"))))
|
|
196
|
+
if binding_type and binding_value is not None and binding_type in allowed:
|
|
197
|
+
record = {"sectionId": section_id, "field": "binding", "value": str(binding_value)}
|
|
198
|
+
checked.append(record)
|
|
199
|
+
if str(binding_value) not in allowed[binding_type]:
|
|
200
|
+
unresolved.append(record)
|
|
201
|
+
return {"schemaVersion": BINDING_SCHEMA, "passed": not unresolved, "checked": checked, "unresolved": unresolved, "referenceTypes": sorted(allowed)}
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def validate_reconcile_contract(contract: object) -> dict[str, Any]:
|
|
205
|
+
errors: list[str] = []
|
|
206
|
+
if not isinstance(contract, dict):
|
|
207
|
+
return {"schemaVersion": IDEMPOTENCY_SCHEMA, "passed": False, "errors": ["reconcile contract must be an object"]}
|
|
208
|
+
if contract.get("schemaVersion") not in {None, IDEMPOTENCY_SCHEMA}:
|
|
209
|
+
errors.append(f"schemaVersion must be {IDEMPOTENCY_SCHEMA}")
|
|
210
|
+
candidate = contract.get("candidateSelection") if isinstance(contract.get("candidateSelection"), dict) else {}
|
|
211
|
+
strategy = str(candidate.get("strategy") or "")
|
|
212
|
+
if strategy not in {"all-candidates", "all_rows", "all-rows"}:
|
|
213
|
+
errors.append("candidateSelection.strategy must select all candidates")
|
|
214
|
+
if candidate.get("filtersOnlyUnconverted") is True:
|
|
215
|
+
errors.append("candidate selection cannot filter only rows that look converted")
|
|
216
|
+
query = str(candidate.get("query") or "").casefold()
|
|
217
|
+
if query and re.search(r"(is\s+null|=\s*'?(empty|converted|none)|jsonb_array_length\s*\([^)]*\)\s*=\s*0)", query):
|
|
218
|
+
errors.append("candidate selection query looks limited to already-unconverted rows")
|
|
219
|
+
compare_fields = contract.get("compareFields")
|
|
220
|
+
if not isinstance(compare_fields, list) or not compare_fields:
|
|
221
|
+
errors.append("compareFields must list every repaired field")
|
|
222
|
+
if contract.get("writesOnlyWhenChanged") is not True:
|
|
223
|
+
errors.append("writesOnlyWhenChanged must be true")
|
|
224
|
+
report = contract.get("report") if isinstance(contract.get("report"), dict) else {}
|
|
225
|
+
for key in ("changed", "unchanged"):
|
|
226
|
+
if report.get(key) is not True:
|
|
227
|
+
errors.append(f"report.{key} evidence is required")
|
|
228
|
+
second = contract.get("secondRun") if isinstance(contract.get("secondRun"), dict) else {}
|
|
229
|
+
if second.get("convergent") is not True:
|
|
230
|
+
errors.append("secondRun.convergent must be true")
|
|
231
|
+
return {"schemaVersion": IDEMPOTENCY_SCHEMA, "passed": not errors, "errors": errors, "mutation": "not executed", "contract": {"candidateStrategy": strategy, "compareFields": compare_fields if isinstance(compare_fields, list) else [], "writesOnlyWhenChanged": contract.get("writesOnlyWhenChanged"), "report": report, "secondRun": second}}
|
|
@@ -193,31 +193,79 @@ def validate_values(registry: dict[str, Any], sections: object) -> dict[str, Any
|
|
|
193
193
|
return {"passed": not errors, "errors": errors}
|
|
194
194
|
|
|
195
195
|
|
|
196
|
-
def
|
|
197
|
-
|
|
196
|
+
def _registry_section_fields(registry: dict[str, Any] | None, section_type: str) -> list[dict[str, Any]]:
|
|
197
|
+
if not isinstance(registry, dict):
|
|
198
|
+
return []
|
|
199
|
+
for section in registry.get("sections", []):
|
|
200
|
+
if isinstance(section, dict) and str(section.get("type")) == section_type:
|
|
201
|
+
return [field for field in section.get("fields", []) if isinstance(field, dict)]
|
|
202
|
+
return []
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def _field_translatable(field: dict[str, Any]) -> bool:
|
|
206
|
+
"""Return the explicit copy policy, defaulting to translatable for text."""
|
|
207
|
+
return field.get("translatable") is not False
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def translation_key_paths(page_id: str, sections: object, field_names: tuple[str, ...] | None = ("imageAlt",), registry: dict[str, Any] | None = None) -> list[str]:
|
|
211
|
+
"""Find stored section fields that need a locale sidecar value.
|
|
212
|
+
|
|
213
|
+
Older callers can continue to pass ``field_names`` and receive the legacy
|
|
214
|
+
image-alt contract. New callers pass a registry, which makes every field
|
|
215
|
+
explicitly marked ``translatable`` part of the locale contract, including
|
|
216
|
+
repeated item fields.
|
|
217
|
+
"""
|
|
198
218
|
if not isinstance(sections, list):
|
|
199
219
|
return []
|
|
200
220
|
paths: list[str] = []
|
|
201
221
|
|
|
202
|
-
def
|
|
222
|
+
def walk_legacy(value: object, path: str) -> None:
|
|
203
223
|
if isinstance(value, dict):
|
|
204
224
|
for key, child in value.items():
|
|
205
225
|
child_path = f"{path}.{key}"
|
|
206
|
-
if key in field_names and isinstance(child, str) and child.strip():
|
|
226
|
+
if field_names and key in field_names and isinstance(child, str) and child.strip():
|
|
207
227
|
paths.append(child_path)
|
|
208
|
-
|
|
228
|
+
walk_legacy(child, child_path)
|
|
209
229
|
elif isinstance(value, list):
|
|
210
230
|
for index, child in enumerate(value):
|
|
211
|
-
|
|
231
|
+
walk_legacy(child, f"{path}.{index}")
|
|
232
|
+
|
|
233
|
+
def walk_declared(value: object, fields: list[dict[str, Any]], path: str) -> None:
|
|
234
|
+
if not isinstance(value, dict):
|
|
235
|
+
return
|
|
236
|
+
for field in fields:
|
|
237
|
+
name = str(field.get("name") or "")
|
|
238
|
+
if not name:
|
|
239
|
+
continue
|
|
240
|
+
child = value.get(name)
|
|
241
|
+
child_path = f"{path}.{name}"
|
|
242
|
+
if _field_translatable(field) and isinstance(child, str) and child.strip():
|
|
243
|
+
paths.append(child_path)
|
|
244
|
+
repeats = field.get("repeats")
|
|
245
|
+
if not isinstance(repeats, dict) or not isinstance(child, list):
|
|
246
|
+
continue
|
|
247
|
+
nested = [item for item in repeats.get("of", []) if isinstance(item, dict)]
|
|
248
|
+
for index, item in enumerate(child):
|
|
249
|
+
item_path = f"{child_path}.{index}"
|
|
250
|
+
if isinstance(item, dict):
|
|
251
|
+
walk_declared(item, nested, item_path)
|
|
252
|
+
elif len(nested) == 1 and _field_translatable(nested[0]) and isinstance(item, str) and item.strip():
|
|
253
|
+
paths.append(item_path)
|
|
212
254
|
|
|
213
255
|
for index, section in enumerate(sections):
|
|
214
|
-
|
|
256
|
+
path = f"page.{page_id}.sections.{section_key(section, index)}"
|
|
257
|
+
if registry and isinstance(section, dict):
|
|
258
|
+
declared = _registry_section_fields(registry, str(section.get("type") or ""))
|
|
259
|
+
if declared:
|
|
260
|
+
walk_declared(section, declared, path)
|
|
261
|
+
continue
|
|
262
|
+
walk_legacy(section, path)
|
|
215
263
|
return paths
|
|
216
264
|
|
|
217
265
|
|
|
218
|
-
def validate_locale_coverage(page_id: str, sections: object, translations: object, locales: Iterable[str]) -> dict[str, Any]:
|
|
266
|
+
def validate_locale_coverage(page_id: str, sections: object, translations: object, locales: Iterable[str], registry: dict[str, Any] | None = None) -> dict[str, Any]:
|
|
219
267
|
"""Require non-default locale rows for every declared translatable field."""
|
|
220
|
-
paths = translation_key_paths(page_id, sections)
|
|
268
|
+
paths = translation_key_paths(page_id, sections, registry=registry)
|
|
221
269
|
values = translations if isinstance(translations, dict) else {}
|
|
222
270
|
missing = [{"locale": locale, "key": key} for locale in locales for key in paths if not isinstance(values.get(locale), dict) or not str(values[locale].get(key) or "").strip()]
|
|
223
271
|
return {"schemaVersion": LOCALE_SCHEMA, "passed": not missing, "pageId": page_id, "requiredKeys": paths, "missing": missing, "writePolicy": "section row and locale rows must be committed in one transaction"}
|
|
@@ -309,12 +357,64 @@ def validate_fanout(registry: dict[str, Any], fanout: object) -> dict[str, Any]:
|
|
|
309
357
|
types = {str(item.get("type")) for item in registry.get("sections", []) if isinstance(item, dict)}
|
|
310
358
|
surfaces = ("typeUnion", "sectionSchema", "blank", "validation", "textExtraction", "renderer", "editor")
|
|
311
359
|
missing: list[dict[str, str]] = []
|
|
360
|
+
|
|
361
|
+
def field_paths(section: dict[str, Any]) -> set[str]:
|
|
362
|
+
result: set[str] = set()
|
|
363
|
+
for field in section.get("fields", []):
|
|
364
|
+
if not isinstance(field, dict) or not field.get("name"):
|
|
365
|
+
continue
|
|
366
|
+
name = str(field["name"])
|
|
367
|
+
result.add(name)
|
|
368
|
+
repeats = field.get("repeats")
|
|
369
|
+
if isinstance(repeats, dict):
|
|
370
|
+
for nested in repeats.get("of", []):
|
|
371
|
+
if isinstance(nested, dict) and nested.get("name"):
|
|
372
|
+
result.add(f"{name}.{nested['name']}")
|
|
373
|
+
return result
|
|
374
|
+
|
|
375
|
+
required_fields = {
|
|
376
|
+
str(section.get("type")): field_paths(section)
|
|
377
|
+
for section in registry.get("sections", [])
|
|
378
|
+
if isinstance(section, dict) and section.get("type")
|
|
379
|
+
}
|
|
380
|
+
|
|
381
|
+
def declared(values: object) -> tuple[set[str], dict[str, set[str]]]:
|
|
382
|
+
if isinstance(values, list):
|
|
383
|
+
return set(map(str, values)), {}
|
|
384
|
+
if not isinstance(values, dict):
|
|
385
|
+
return set(), {}
|
|
386
|
+
types = set(map(str, values.keys()))
|
|
387
|
+
fields: dict[str, set[str]] = {}
|
|
388
|
+
for section_type, entry in values.items():
|
|
389
|
+
if isinstance(entry, list):
|
|
390
|
+
fields[str(section_type)] = set(map(str, entry))
|
|
391
|
+
elif isinstance(entry, dict):
|
|
392
|
+
candidate = entry.get("fields", [])
|
|
393
|
+
if isinstance(candidate, list):
|
|
394
|
+
fields[str(section_type)] = set(map(str, candidate))
|
|
395
|
+
return types, fields
|
|
396
|
+
|
|
312
397
|
for surface in surfaces:
|
|
313
398
|
values = fanout.get(surface)
|
|
314
|
-
|
|
315
|
-
for section_type in sorted(types -
|
|
399
|
+
declared_types, declared_fields = declared(values)
|
|
400
|
+
for section_type in sorted(types - declared_types):
|
|
316
401
|
missing.append({"type": section_type, "surface": surface})
|
|
317
|
-
|
|
402
|
+
for section_type in sorted(types & declared_types):
|
|
403
|
+
for field in sorted(required_fields.get(section_type, set()) - declared_fields.get(section_type, set())):
|
|
404
|
+
missing.append({"type": section_type, "surface": f"{surface}.{field}"})
|
|
405
|
+
|
|
406
|
+
editor_screens = fanout.get("editorScreens")
|
|
407
|
+
if not isinstance(editor_screens, dict) or len(editor_screens) < 2:
|
|
408
|
+
missing.append({"type": "editor", "surface": "editorScreens (at least two screens required)"})
|
|
409
|
+
elif isinstance(editor_screens, dict):
|
|
410
|
+
for screen, values in editor_screens.items():
|
|
411
|
+
screen_types, screen_fields = declared(values)
|
|
412
|
+
for section_type in sorted(types - screen_types):
|
|
413
|
+
missing.append({"type": section_type, "surface": f"editorScreens.{screen}"})
|
|
414
|
+
for section_type in sorted(types & screen_types):
|
|
415
|
+
for field in sorted(required_fields.get(section_type, set()) - screen_fields.get(section_type, set())):
|
|
416
|
+
missing.append({"type": section_type, "surface": f"editorScreens.{screen}.{field}"})
|
|
417
|
+
return {"schemaVersion": FANOUT_SCHEMA, "passed": not missing, "errors": [f"{item['type']} missing {item['surface']} fan-out" for item in missing], "sectionTypes": sorted(types), "missing": missing, "fieldAware": True, "editorScreenCount": len(editor_screens) if isinstance(editor_screens, dict) else 0}
|
|
318
418
|
|
|
319
419
|
|
|
320
420
|
def reconcile_fields(current: object, desired: object) -> dict[str, Any]:
|
|
@@ -14,6 +14,7 @@ HARD_URL_LIMIT = 50_000
|
|
|
14
14
|
HARD_BYTES_LIMIT = 52_428_800
|
|
15
15
|
DEFAULT_CHUNK_TARGET = 500
|
|
16
16
|
EXCLUDED_PATHS = re.compile(r"/(search|find|login|draft|preview)(/|$)", re.I)
|
|
17
|
+
W3C_UTC_DATETIME = re.compile(r"^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}Z$")
|
|
17
18
|
|
|
18
19
|
|
|
19
20
|
def absolute_url(url: str, origin: str) -> bool:
|
|
@@ -121,7 +122,7 @@ def build_plan(routes: list[dict[str, str]], origin: str, content_types: set[str
|
|
|
121
122
|
current = set(sitemap_urls)
|
|
122
123
|
old = set((previous or {}).get("sitemapUrls", []))
|
|
123
124
|
redirects = [{"from": url, "to": origin.rstrip("/") + "/sitemap.xml", "reason": "previously advertised sitemap removed"} for url in sorted(old - current)]
|
|
124
|
-
plan = {"schemaVersion": "maggie-seo-sitemap-plan.v1", "planId": "sitemap-" + datetime.now(timezone.utc).strftime("%Y%m%d%H%M%S"), "origin": origin.rstrip("/"), "groups": group_plans, "index": {"url": origin.rstrip("/") + "/sitemap.xml", "bytes": len(index_xml.encode()), "xml": index_xml, "sitemapUrls": sitemap_urls, "responseHeaders": {"content-type": "text/xml; charset=utf-8", "x-robots-tag": "
|
|
125
|
+
plan = {"schemaVersion": "maggie-seo-sitemap-plan.v1", "planId": "sitemap-" + datetime.now(timezone.utc).strftime("%Y%m%d%H%M%S"), "origin": origin.rstrip("/"), "groups": group_plans, "index": {"url": origin.rstrip("/") + "/sitemap.xml", "bytes": len(index_xml.encode()), "xml": index_xml, "sitemapUrls": sitemap_urls, "responseHeaders": {"content-type": "text/xml; charset=utf-8", "x-robots-tag": "noindex"}}, "redirects": redirects, "excluded": excluded}
|
|
125
126
|
plan["validation"] = validate_plan_data(plan)
|
|
126
127
|
return plan
|
|
127
128
|
|
|
@@ -163,9 +164,16 @@ def validate_plan_data(plan: dict, strict_semantic: bool = False) -> dict:
|
|
|
163
164
|
for link in root.iter():
|
|
164
165
|
if link.tag.endswith("link") and link.get("rel") == "alternate" and not absolute_url(link.get("href", ""), origin):
|
|
165
166
|
errors.append(f"chunk contains a relative or off-origin alternate: {chunk.get('filename')}")
|
|
167
|
+
if link.tag.endswith("lastmod") and not W3C_UTC_DATETIME.fullmatch((link.text or "").strip()):
|
|
168
|
+
errors.append(f"lastmod must be a full W3C UTC datetime: {chunk.get('filename')}")
|
|
166
169
|
except ElementTree.ParseError:
|
|
167
170
|
errors.append(f"chunk is not valid XML: {chunk.get('filename')}")
|
|
168
171
|
index = plan.get("index", {})
|
|
172
|
+
headers = index.get("responseHeaders") if isinstance(index.get("responseHeaders"), dict) else {}
|
|
173
|
+
if headers.get("content-type") != "text/xml; charset=utf-8":
|
|
174
|
+
errors.append("sitemap response content-type must be text/xml; charset=utf-8")
|
|
175
|
+
if str(headers.get("x-robots-tag") or "").casefold() != "noindex":
|
|
176
|
+
errors.append("sitemap response x-robots-tag must be noindex")
|
|
169
177
|
if index.get("bytes", 0) > HARD_BYTES_LIMIT:
|
|
170
178
|
errors.append("sitemap index exceeds byte limit")
|
|
171
179
|
try:
|