@topy-ai/maggie 0.7.11 → 0.7.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +20 -8
- package/README.zh-TW.md +12 -1
- package/bin/maggie.js +4 -2
- package/bundled-contracts/maggiedash/README.md +3 -0
- package/bundled-contracts/maggiedash/quality-contracts.md +58 -0
- package/bundled-skills/maggie-dash/SKILL.md +31 -8
- package/bundled-skills/maggie-deployment/SKILL.md +15 -0
- package/bundled-skills/maggie-seo-geo/SKILL.md +20 -0
- package/bundled-templates/maggiedash/section-fanout.json +84 -8
- package/bundled-templates/maggiedash/section-registry.json +3 -3
- package/bundled-tools/clis/maggie_dash.py +48 -2
- package/bundled-tools/clis/maggie_migration.py +67 -0
- package/bundled-tools/clis/site_audit.py +17 -0
- package/bundled-tools/runtime/maggie_quality.py +231 -0
- package/bundled-tools/runtime/maggie_sections.py +112 -12
- package/bundled-tools/runtime/maggie_sitemap.py +31 -6
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -104,7 +104,8 @@ maggie bootstrap interview | phase ...
|
|
|
104
104
|
maggie dash install | init | status | migrate | cms ...
|
|
105
105
|
maggie dash transition ... # explicit content approval transition
|
|
106
106
|
maggie dash variant ... # service variant create/review/preview/publish
|
|
107
|
-
maggie dash sections ... #
|
|
107
|
+
maggie dash sections ... # field fan-out, locale, binding, media, copy, identity keys
|
|
108
|
+
maggie dash inventory ... # disjoint published-page inventory
|
|
108
109
|
maggie agent-content write ... # host-authorized, origin-bound content bridge
|
|
109
110
|
maggie verification coverage ... # changed surface/locale evidence gate
|
|
110
111
|
maggie clone ... # authorized homepage capture
|
|
@@ -119,6 +120,7 @@ maggie seo performance ... # sampled PageSpeed/CWV report and baseline
|
|
|
119
120
|
maggie seo images ... # inventory, variants, confirmation, validate
|
|
120
121
|
maggie seo sitemap ... # typed/semantic plan, agent-files, apply, rollback
|
|
121
122
|
maggie deployment | migration | release | analytics | schedule
|
|
123
|
+
maggie migration identity --identity-file FILE [--expected-file FILE]
|
|
122
124
|
maggie deployment canary --asset URL=SHA256 --render-report report.json
|
|
123
125
|
maggie design icon-inventory --source-dir src --runtime assets/icons.css
|
|
124
126
|
maggie api lifecycle | site-audit | ops audit
|
|
@@ -218,8 +220,8 @@ artifact schemas.
|
|
|
218
220
|
Recommended upgrade sequence for the current release:
|
|
219
221
|
|
|
220
222
|
```bash
|
|
221
|
-
npx @topy-ai/maggie@0.7.
|
|
222
|
-
npx @topy-ai/maggie@0.7.
|
|
223
|
+
npx @topy-ai/maggie@0.7.13 update --project . --force
|
|
224
|
+
npx @topy-ai/maggie@0.7.13 cleanup --project .
|
|
223
225
|
```
|
|
224
226
|
|
|
225
227
|
Maintainers should pass npm credentials through the repository helper, never
|
|
@@ -229,7 +231,11 @@ as a command-line argument:
|
|
|
229
231
|
node scripts/publish-npm.mjs --maggie-env-file ../.env
|
|
230
232
|
```
|
|
231
233
|
|
|
232
|
-
The 0.7.
|
|
234
|
+
The 0.7.13 workflow adds field-aware section fan-out and locale coverage,
|
|
235
|
+
sibling-copy/media checks, disjoint page inventory, binding validation,
|
|
236
|
+
idempotency and database-target identity gates, full W3C sitemap lastmod
|
|
237
|
+
validation, and the accepted `X-Robots-Tag: noindex` response contract. The
|
|
238
|
+
0.7.12 workflow adds served-content equivalence checks, query-route
|
|
233
239
|
baseline exclusions, changed-surface render evidence gates, generated skill
|
|
234
240
|
catalogs, component-binding audits, and icon-family noise filtering. It also
|
|
235
241
|
includes the 0.7.9 nested section-field contracts, renderer-backed examples,
|
|
@@ -255,7 +261,9 @@ also adds image-aware content equivalence with explicit legacy-baseline
|
|
|
255
261
|
limitations, locale checks for stored image alt text, a validated label/value
|
|
256
262
|
`pairs` section, registry fan-out checks, idempotent reconciliation, per-step
|
|
257
263
|
and section-scoped design evidence, catch-all source confidence, shell-safe VPS
|
|
258
|
-
SSH guidance,
|
|
264
|
+
SSH guidance, compatible feedback CLI examples, conventional sitemap
|
|
265
|
+
formatting, response-header guidance, and access-log diagnostics for absent
|
|
266
|
+
crawler requests. The release
|
|
259
267
|
retains the existing service matching,
|
|
260
268
|
localization, seed-manifest, lockfile/analytics, sitemap, deployment and
|
|
261
269
|
rollback workflows.
|
|
@@ -512,9 +520,13 @@ maggie design section-validate --before before.json --after after.json \
|
|
|
512
520
|
```
|
|
513
521
|
|
|
514
522
|
MaggieDash section contracts include a compact `pairs` band for factual
|
|
515
|
-
label/value rows. Validate required values, registry fan-out,
|
|
516
|
-
|
|
517
|
-
`
|
|
523
|
+
label/value rows. Validate required values, field-aware registry fan-out,
|
|
524
|
+
all locale copy fields, sibling copy/media, stored bindings, and idempotent
|
|
525
|
+
repairs with `maggie dash sections validate`, `fanout-validate`,
|
|
526
|
+
`locale-validate`, `variant-copy-validate`, `media-validate`,
|
|
527
|
+
`bindings-validate`, `idempotency-validate`, and `reconcile`. Use
|
|
528
|
+
`maggie dash inventory` to classify published pages once into disjoint kinds.
|
|
529
|
+
See the [quality contract examples](https://github.com/TOPY-AI-LTD/ai-cmo-skills/blob/main/contracts/maggiedash/quality-contracts.md).
|
|
518
530
|
|
|
519
531
|
The package includes all 18 installable skills: `maggie-blog-bootstrap`,
|
|
520
532
|
`maggie-dash`, `maggie-clone`, `maggie-clone-to-template`, `maggie-marketplace`,
|
package/README.zh-TW.md
CHANGED
|
@@ -8,7 +8,7 @@ Codex、Claude Code 與相容的 coding agents。
|
|
|
8
8
|
## 安裝
|
|
9
9
|
|
|
10
10
|
```bash
|
|
11
|
-
npx @topy-ai/maggie@0.7.
|
|
11
|
+
npx @topy-ai/maggie@0.7.13 init --agent all
|
|
12
12
|
npx @topy-ai/maggie doctor --project .
|
|
13
13
|
```
|
|
14
14
|
|
|
@@ -39,6 +39,17 @@ maggie deployment canary --project . \
|
|
|
39
39
|
--asset https://example.com/assets/app.js=<sha256> \
|
|
40
40
|
--render-report .maggie/rendered-canary.json \
|
|
41
41
|
--output docs/deployment-canary.json
|
|
42
|
+
|
|
43
|
+
# MaggieDash section/page quality contracts
|
|
44
|
+
maggie dash sections fanout-validate --registry templates/maggiedash/section-registry.json \
|
|
45
|
+
--fanout-file templates/maggiedash/section-fanout.json
|
|
46
|
+
maggie dash sections variant-copy-validate --pages-file .maggie/variant-pages.json
|
|
47
|
+
maggie dash sections media-validate --pages-file .maggie/variant-pages.json --across-siblings
|
|
48
|
+
maggie dash inventory --pages-file .maggie/published-pages.json
|
|
49
|
+
|
|
50
|
+
# Migration target identity (never prints or stores a database URL)
|
|
51
|
+
maggie migration identity --identity-file .maggie/db-identity.json \
|
|
52
|
+
--expected-file .maggie/service-db-identity.json
|
|
42
53
|
```
|
|
43
54
|
|
|
44
55
|
完整中文說明、18 個 skills 清單和 roadmap:
|
package/bin/maggie.js
CHANGED
|
@@ -58,7 +58,8 @@ Usage:
|
|
|
58
58
|
maggie doctor [--project PATH]
|
|
59
59
|
maggie bootstrap interview [project]
|
|
60
60
|
maggie dash init|install|status|migrate|transition|variant|cms --project PATH [options]
|
|
61
|
-
maggie dash sections <validate|prompt|keys|remap-translations|fanout-validate|locale-validate|reconcile> [options]
|
|
61
|
+
maggie dash sections <validate|prompt|keys|remap-translations|fanout-validate|locale-validate|variant-copy-validate|media-validate|bindings-validate|idempotency-validate|reconcile> [options]
|
|
62
|
+
maggie dash inventory --pages-file FILE
|
|
62
63
|
maggie dash components-audit --bindings-file FILE --sections-file FILE --pages-file FILE
|
|
63
64
|
maggie dash status --project PATH
|
|
64
65
|
maggie dash migrate --project PATH --confirm
|
|
@@ -104,6 +105,7 @@ Usage:
|
|
|
104
105
|
maggie deployment --project PATH --target vps-with-cloudflare-dns
|
|
105
106
|
maggie deployment canary --project PATH --asset URL=SHA256 --render-report report.json --output docs/deployment-canary.json
|
|
106
107
|
maggie migration --project PATH --environment staging
|
|
108
|
+
maggie migration identity --identity-file FILE [--expected-file FILE]
|
|
107
109
|
maggie schedule PATH/.maggie/schedule.json --project PATH
|
|
108
110
|
maggie analytics [traffic-audit|release-gate] --project PATH --environment staging
|
|
109
111
|
maggie release PATH --environment staging --target vps-with-cloudflare-dns
|
|
@@ -112,7 +114,7 @@ Usage:
|
|
|
112
114
|
maggie localization <extract|plan|generate|preview|validate|review|publish|stale|glossary> [options]
|
|
113
115
|
maggie seo performance|images|sitemap [options] (sitemap supports strict validate and agent-files)
|
|
114
116
|
maggie feedback <collect|preview|submit|list> [options]
|
|
115
|
-
maggie site-audit URL [--crawl] [--languages en-GB,es-MX,ja-JP] [--check-hreflang]
|
|
117
|
+
maggie site-audit URL [--crawl] [--access-log FILE] [--require-sitemap-request] [--languages en-GB,es-MX,ja-JP] [--check-hreflang]
|
|
116
118
|
maggie site-audit URL --crawl --save-baseline FILE --reviewer NAME
|
|
117
119
|
maggie site-audit URL --crawl --baseline FILE
|
|
118
120
|
maggie browser-audit URL --browse PATH --output DIR --required SELECTOR [--sticky SELECTOR]
|
|
@@ -24,6 +24,9 @@ frontend framework.
|
|
|
24
24
|
for translation keys and ordered migration;
|
|
25
25
|
- [`translation-cache-policy-v1.json`](translation-cache-policy-v1.json):
|
|
26
26
|
restart-after-out-of-band-write evidence.
|
|
27
|
+
- [`quality-contracts.md`](quality-contracts.md): reviewable inputs for
|
|
28
|
+
disjoint page inventory, variant/media checks, binding resolution, repair
|
|
29
|
+
convergence, and database target identity.
|
|
27
30
|
|
|
28
31
|
The section registry is an adapter input rather than a fixed six-band list.
|
|
29
32
|
Each registry entry must describe its purpose and limits, include a
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
# MaggieDash quality contracts
|
|
2
|
+
|
|
3
|
+
These inputs are reviewable JSON evidence. They do not write a database or
|
|
4
|
+
publish content.
|
|
5
|
+
|
|
6
|
+
## Page inventory
|
|
7
|
+
|
|
8
|
+
```json
|
|
9
|
+
{
|
|
10
|
+
"pages": [
|
|
11
|
+
{"path": "/", "sections": [{"type": "hero"}]},
|
|
12
|
+
{"path": "/about", "source": "src/pages/about.astro"}
|
|
13
|
+
]
|
|
14
|
+
}
|
|
15
|
+
```
|
|
16
|
+
|
|
17
|
+
Run `maggie dash inventory --pages-file .maggie/published-pages.json`. The
|
|
18
|
+
result assigns each path exactly one kind and reports `bandCount` and
|
|
19
|
+
`codeRenderedCount`; unknown or duplicate paths fail.
|
|
20
|
+
|
|
21
|
+
## Variant and media checks
|
|
22
|
+
|
|
23
|
+
Variant pages should provide `id`, `family`, `locale`, and either a `sections`
|
|
24
|
+
array or direct fields. Run `variant-copy-validate` with one or more copy field
|
|
25
|
+
names (the default is `faq`). Run `media-validate` for page duplicates and add
|
|
26
|
+
`--across-siblings` to catch shared media in one variant family.
|
|
27
|
+
|
|
28
|
+
## Bindings
|
|
29
|
+
|
|
30
|
+
The sections file may contain a `binding` object or fields such as
|
|
31
|
+
`categorySlug`, `collectionId`, or `serviceRef`. The references file maps each
|
|
32
|
+
type to IDs, slugs, names, or values. `bindings-validate` fails when a stored
|
|
33
|
+
reference cannot be resolved.
|
|
34
|
+
|
|
35
|
+
## Repair idempotency
|
|
36
|
+
|
|
37
|
+
```json
|
|
38
|
+
{
|
|
39
|
+
"schemaVersion": "maggie-reconcile-contract.v1",
|
|
40
|
+
"candidateSelection": {"strategy": "all-candidates", "query": "SELECT all repair candidates"},
|
|
41
|
+
"compareFields": ["sections", "translations"],
|
|
42
|
+
"writesOnlyWhenChanged": true,
|
|
43
|
+
"report": {"changed": true, "unchanged": true},
|
|
44
|
+
"secondRun": {"convergent": true}
|
|
45
|
+
}
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
`idempotency-validate` rejects a selection that only targets rows that look
|
|
49
|
+
unconverted. The actual adapter must still use a transaction and report its
|
|
50
|
+
real changed/unchanged counts.
|
|
51
|
+
|
|
52
|
+
## Database identity
|
|
53
|
+
|
|
54
|
+
`maggie migration identity` accepts a `maggie-database-identity.v1` report with
|
|
55
|
+
environment, target/database name, server identity, service release identity,
|
|
56
|
+
row counts, and a matching SHA-256 fingerprint. It can compare a second
|
|
57
|
+
service-owned report. Reports must never contain a database URL, password, or
|
|
58
|
+
credential.
|
|
@@ -182,15 +182,26 @@ maggie dash sections remap-translations \
|
|
|
182
182
|
--mapping-file .maggie/section-translation-map.json \
|
|
183
183
|
--output .maggie/translations-v2.json --confirm
|
|
184
184
|
|
|
185
|
-
# Validate
|
|
185
|
+
# Validate every registry-declared translatable field for every locale before publishing:
|
|
186
186
|
maggie dash sections locale-validate \
|
|
187
187
|
--page-id <page-id> --sections-file .maggie/sections.json \
|
|
188
|
-
--translations-file .maggie/translations-by-locale.json
|
|
188
|
+
--translations-file .maggie/translations-by-locale.json \
|
|
189
|
+
--registry templates/maggiedash/section-registry.json --locale zh-Hant
|
|
189
190
|
|
|
190
191
|
# Check that a host wired every registry type through all implementation surfaces:
|
|
191
192
|
maggie dash sections fanout-validate \
|
|
192
193
|
--registry templates/maggiedash/section-registry.json \
|
|
193
194
|
--fanout-file templates/maggiedash/section-fanout.json
|
|
195
|
+
|
|
196
|
+
# Validate sibling copy, media uniqueness, references, and repair convergence:
|
|
197
|
+
maggie dash sections variant-copy-validate --pages-file .maggie/variant-pages.json --field faq
|
|
198
|
+
maggie dash sections media-validate --pages-file .maggie/variant-pages.json --across-siblings
|
|
199
|
+
maggie dash sections bindings-validate --sections-file .maggie/sections.json \
|
|
200
|
+
--references-file .maggie/reference-inventory.json
|
|
201
|
+
maggie dash sections idempotency-validate --contract-file .maggie/reconcile-contract.json
|
|
202
|
+
|
|
203
|
+
# Classify every published page exactly once:
|
|
204
|
+
maggie dash inventory --pages-file .maggie/published-pages.json
|
|
194
205
|
```
|
|
195
206
|
|
|
196
207
|
The catalogue declares purpose, usage, placement, repeatability and layout
|
|
@@ -211,10 +222,13 @@ as a fixed number of bands.
|
|
|
211
222
|
Required fields and repeated `pairs` values cannot be blank. The `pairs`
|
|
212
223
|
starter band is intended for compact label/value facts such as opening hours;
|
|
213
224
|
it is not a prose fallback. The registry fan-out manifest is a host integration
|
|
214
|
-
contract: when a host adds a type, update its type union/schema, blank
|
|
215
|
-
validation, readable-content extraction, renderer, and editor entry
|
|
216
|
-
then run `fanout-validate`.
|
|
217
|
-
|
|
225
|
+
contract: when a host adds a type or field, update its type union/schema, blank
|
|
226
|
+
state, validation, readable-content extraction, renderer, and editor entry
|
|
227
|
+
together, then run `fanout-validate`. The check is field-aware and requires two
|
|
228
|
+
editor screens, so adding a field without wiring both editing surfaces fails
|
|
229
|
+
closed. `locale-validate --registry` checks every declared translatable scalar
|
|
230
|
+
and repeated item field; URLs and media sources can be marked non-translatable.
|
|
231
|
+
The host adapter must write the section and all non-default locale rows in one
|
|
218
232
|
transaction. The JSON report is evidence, not a database write.
|
|
219
233
|
|
|
220
234
|
Translation keys use stable section IDs, with an array-index fallback only
|
|
@@ -233,8 +247,17 @@ is the reviewable output form, not a substitute for the adapter's transaction.
|
|
|
233
247
|
Keep data-owned values out of page copy. For example, a pricing band should
|
|
234
248
|
store a service identifier or slug and let the live service/booking adapter
|
|
235
249
|
render current options and prices. Inventory reports must classify each
|
|
236
|
-
published page once into disjoint groups;
|
|
237
|
-
|
|
250
|
+
published page once into disjoint groups; `maggie dash inventory` reports band
|
|
251
|
+
and code-rendered counts. Price options, arrangements, and other child records
|
|
252
|
+
are counts, not pages. Run the variant-copy and media checks before publishing
|
|
253
|
+
localized sibling pages, and the bindings check whenever a section stores a
|
|
254
|
+
category, collection, or service reference; an unresolved binding is a publish
|
|
255
|
+
failure, not an empty state.
|
|
256
|
+
|
|
257
|
+
Repair scripts must select all candidates, compare every target field, write
|
|
258
|
+
only changed rows, and report both changed and unchanged rows. Validate that
|
|
259
|
+
contract with `idempotency-validate`; a query that selects only rows that look
|
|
260
|
+
unconverted cannot repair its own bad output.
|
|
238
261
|
|
|
239
262
|
After any script or direct adapter write to translation data, invalidate the
|
|
240
263
|
running process before verification. Record the restart and run the rendering
|
|
@@ -111,6 +111,21 @@ The manifest must reference `DATABASE_URL` (not a literal URL), declare an
|
|
|
111
111
|
advancing version, forward-only/idempotent policy, backup and restore commands,
|
|
112
112
|
and a tested restore artifact. The validator never runs those commands.
|
|
113
113
|
|
|
114
|
+
Before a write, prove that the migration target is the database used by the
|
|
115
|
+
running service. The identity report is intentionally secret-free and includes
|
|
116
|
+
the environment, target/database name, server identity, service release
|
|
117
|
+
identity, row counts, and a SHA-256 fingerprint. Compare it with the deployment
|
|
118
|
+
identity when available:
|
|
119
|
+
|
|
120
|
+
```bash
|
|
121
|
+
maggie migration identity --identity-file .maggie/db-identity.json \
|
|
122
|
+
--expected-file .maggie/service-db-identity.json
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
The gate fails on a missing or mismatched target identity and never executes a
|
|
126
|
+
migration. A successful connection alone is not proof that the intended
|
|
127
|
+
database was selected.
|
|
128
|
+
|
|
114
129
|
If a release depends on existing rows or seeded data, set
|
|
115
130
|
`dataDependencies: true` in `.maggie/migration-manifest.json` and provide
|
|
116
131
|
`.maggie/deployment/data-release.json` before deployment. The checkpoint must
|
|
@@ -59,6 +59,26 @@ If the change date is unknown, omit `lastmod`. An empty content-type does not
|
|
|
59
59
|
need a sitemap chunk in the sitemap index; serving an empty endpoint and
|
|
60
60
|
advertising it are separate decisions.
|
|
61
61
|
|
|
62
|
+
Generated sitemap XML uses the conventional readable shape by default: one
|
|
63
|
+
`<url>`/`<sitemap>` entry per block, UTF-8 XML, and date-only `lastmod` evidence
|
|
64
|
+
rendered as a full UTC W3C datetime. The plan also exposes the response
|
|
65
|
+
contract (`text/xml; charset=utf-8` and `X-Robots-Tag: noindex`) for the framework
|
|
66
|
+
route or reverse proxy to apply; writing a file cannot set HTTP headers itself.
|
|
67
|
+
When Search Console reports a valid sitemap as unfetched, probe success alone
|
|
68
|
+
is not a diagnosis. Supply a sanitized local access log to distinguish
|
|
69
|
+
`googlebot-request-observed`, another-client-only traffic, and
|
|
70
|
+
`no-matching-request`:
|
|
71
|
+
|
|
72
|
+
```bash
|
|
73
|
+
maggie site-audit https://example.com --access-log /var/log/nginx/access.log
|
|
74
|
+
maggie site-audit https://example.com --access-log /var/log/nginx/access.log \
|
|
75
|
+
--require-sitemap-request
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
The strict flag is for a review window where a crawler request is expected; it
|
|
79
|
+
fails only when the supplied log contains no sitemap request. Never upload raw
|
|
80
|
+
logs or include IPs, credentials, or query data in feedback or reports.
|
|
81
|
+
|
|
62
82
|
## Freeze and compare a reviewed site
|
|
63
83
|
|
|
64
84
|
```bash
|
|
@@ -1,11 +1,87 @@
|
|
|
1
1
|
{
|
|
2
2
|
"schemaVersion": "maggiedash-section-fanout.v1",
|
|
3
|
-
"description": "
|
|
4
|
-
"typeUnion":
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
3
|
+
"description": "Field-aware host contract: every registry type and declared field must be wired through each implementation surface and two editor screens.",
|
|
4
|
+
"typeUnion": {
|
|
5
|
+
"hero": ["title", "intro", "image", "imageAlt", "ctaLabel"],
|
|
6
|
+
"prose": ["heading", "paragraphs", "paragraphs.paragraph", "image", "imageAlt"],
|
|
7
|
+
"pairs": ["heading", "rows", "rows.label", "rows.value", "note"],
|
|
8
|
+
"cards": ["heading", "items", "items.title", "items.body"],
|
|
9
|
+
"links": ["heading", "items", "items.label", "items.href", "items.body"],
|
|
10
|
+
"faq": ["heading", "items", "items.question", "items.answer"],
|
|
11
|
+
"cta": ["heading", "body", "label"]
|
|
12
|
+
},
|
|
13
|
+
"sectionSchema": {
|
|
14
|
+
"hero": ["title", "intro", "image", "imageAlt", "ctaLabel"],
|
|
15
|
+
"prose": ["heading", "paragraphs", "paragraphs.paragraph", "image", "imageAlt"],
|
|
16
|
+
"pairs": ["heading", "rows", "rows.label", "rows.value", "note"],
|
|
17
|
+
"cards": ["heading", "items", "items.title", "items.body"],
|
|
18
|
+
"links": ["heading", "items", "items.label", "items.href", "items.body"],
|
|
19
|
+
"faq": ["heading", "items", "items.question", "items.answer"],
|
|
20
|
+
"cta": ["heading", "body", "label"]
|
|
21
|
+
},
|
|
22
|
+
"blank": {
|
|
23
|
+
"hero": ["title", "intro", "image", "imageAlt", "ctaLabel"],
|
|
24
|
+
"prose": ["heading", "paragraphs", "paragraphs.paragraph", "image", "imageAlt"],
|
|
25
|
+
"pairs": ["heading", "rows", "rows.label", "rows.value", "note"],
|
|
26
|
+
"cards": ["heading", "items", "items.title", "items.body"],
|
|
27
|
+
"links": ["heading", "items", "items.label", "items.href", "items.body"],
|
|
28
|
+
"faq": ["heading", "items", "items.question", "items.answer"],
|
|
29
|
+
"cta": ["heading", "body", "label"]
|
|
30
|
+
},
|
|
31
|
+
"validation": {
|
|
32
|
+
"hero": ["title", "intro", "image", "imageAlt", "ctaLabel"],
|
|
33
|
+
"prose": ["heading", "paragraphs", "paragraphs.paragraph", "image", "imageAlt"],
|
|
34
|
+
"pairs": ["heading", "rows", "rows.label", "rows.value", "note"],
|
|
35
|
+
"cards": ["heading", "items", "items.title", "items.body"],
|
|
36
|
+
"links": ["heading", "items", "items.label", "items.href", "items.body"],
|
|
37
|
+
"faq": ["heading", "items", "items.question", "items.answer"],
|
|
38
|
+
"cta": ["heading", "body", "label"]
|
|
39
|
+
},
|
|
40
|
+
"textExtraction": {
|
|
41
|
+
"hero": ["title", "intro", "image", "imageAlt", "ctaLabel"],
|
|
42
|
+
"prose": ["heading", "paragraphs", "paragraphs.paragraph", "image", "imageAlt"],
|
|
43
|
+
"pairs": ["heading", "rows", "rows.label", "rows.value", "note"],
|
|
44
|
+
"cards": ["heading", "items", "items.title", "items.body"],
|
|
45
|
+
"links": ["heading", "items", "items.label", "items.href", "items.body"],
|
|
46
|
+
"faq": ["heading", "items", "items.question", "items.answer"],
|
|
47
|
+
"cta": ["heading", "body", "label"]
|
|
48
|
+
},
|
|
49
|
+
"renderer": {
|
|
50
|
+
"hero": ["title", "intro", "image", "imageAlt", "ctaLabel"],
|
|
51
|
+
"prose": ["heading", "paragraphs", "paragraphs.paragraph", "image", "imageAlt"],
|
|
52
|
+
"pairs": ["heading", "rows", "rows.label", "rows.value", "note"],
|
|
53
|
+
"cards": ["heading", "items", "items.title", "items.body"],
|
|
54
|
+
"links": ["heading", "items", "items.label", "items.href", "items.body"],
|
|
55
|
+
"faq": ["heading", "items", "items.question", "items.answer"],
|
|
56
|
+
"cta": ["heading", "body", "label"]
|
|
57
|
+
},
|
|
58
|
+
"editor": {
|
|
59
|
+
"hero": ["title", "intro", "image", "imageAlt", "ctaLabel"],
|
|
60
|
+
"prose": ["heading", "paragraphs", "paragraphs.paragraph", "image", "imageAlt"],
|
|
61
|
+
"pairs": ["heading", "rows", "rows.label", "rows.value", "note"],
|
|
62
|
+
"cards": ["heading", "items", "items.title", "items.body"],
|
|
63
|
+
"links": ["heading", "items", "items.label", "items.href", "items.body"],
|
|
64
|
+
"faq": ["heading", "items", "items.question", "items.answer"],
|
|
65
|
+
"cta": ["heading", "body", "label"]
|
|
66
|
+
},
|
|
67
|
+
"editorScreens": {
|
|
68
|
+
"section-editor": {
|
|
69
|
+
"hero": ["title", "intro", "image", "imageAlt", "ctaLabel"],
|
|
70
|
+
"prose": ["heading", "paragraphs", "paragraphs.paragraph", "image", "imageAlt"],
|
|
71
|
+
"pairs": ["heading", "rows", "rows.label", "rows.value", "note"],
|
|
72
|
+
"cards": ["heading", "items", "items.title", "items.body"],
|
|
73
|
+
"links": ["heading", "items", "items.label", "items.href", "items.body"],
|
|
74
|
+
"faq": ["heading", "items", "items.question", "items.answer"],
|
|
75
|
+
"cta": ["heading", "body", "label"]
|
|
76
|
+
},
|
|
77
|
+
"document-detail": {
|
|
78
|
+
"hero": ["title", "intro", "image", "imageAlt", "ctaLabel"],
|
|
79
|
+
"prose": ["heading", "paragraphs", "paragraphs.paragraph", "image", "imageAlt"],
|
|
80
|
+
"pairs": ["heading", "rows", "rows.label", "rows.value", "note"],
|
|
81
|
+
"cards": ["heading", "items", "items.title", "items.body"],
|
|
82
|
+
"links": ["heading", "items", "items.label", "items.href", "items.body"],
|
|
83
|
+
"faq": ["heading", "items", "items.question", "items.answer"],
|
|
84
|
+
"cta": ["heading", "body", "label"]
|
|
85
|
+
}
|
|
86
|
+
}
|
|
11
87
|
}
|
|
@@ -13,7 +13,7 @@
|
|
|
13
13
|
"fields": [
|
|
14
14
|
{"name": "title", "usage": "Plain words a visitor would search for.", "limit": 70, "required": true},
|
|
15
15
|
{"name": "intro", "usage": "Who it is for and what it does.", "limit": 220},
|
|
16
|
-
{"name": "image", "usage": "The hero image rendered beside or behind the copy.", "limit": 500},
|
|
16
|
+
{"name": "image", "usage": "The hero image rendered beside or behind the copy.", "limit": 500, "translatable": false},
|
|
17
17
|
{"name": "imageAlt", "usage": "What the hero image shows for a reader who cannot see it.", "limit": 160},
|
|
18
18
|
{"name": "ctaLabel", "usage": "The action as a verb.", "limit": 30}
|
|
19
19
|
],
|
|
@@ -28,7 +28,7 @@
|
|
|
28
28
|
"fields": [
|
|
29
29
|
{"name": "heading", "usage": "The argument in a sentence.", "limit": 90},
|
|
30
30
|
{"name": "paragraphs", "usage": "Standalone paragraphs.", "limit": 0, "repeats": {"min": 1, "max": 4, "of": [{"name": "paragraph", "usage": "One paragraph of plain prose.", "limit": 400}]}},
|
|
31
|
-
{"name": "image", "usage": "An optional supporting image.", "limit": 500},
|
|
31
|
+
{"name": "image", "usage": "An optional supporting image.", "limit": 500, "translatable": false},
|
|
32
32
|
{"name": "imageAlt", "usage": "What the supporting image shows.", "limit": 160}
|
|
33
33
|
],
|
|
34
34
|
"example": {"type": "prose", "heading": "Make the important part easier to understand", "paragraphs": ["Start with the decision your reader is trying to make.", "Then give them the context, evidence and next step in that order."]}
|
|
@@ -66,7 +66,7 @@
|
|
|
66
66
|
"repeatable": true,
|
|
67
67
|
"fields": [
|
|
68
68
|
{"name": "heading", "usage": "What the links have in common.", "limit": 90},
|
|
69
|
-
{"name": "items", "usage": "Real related pages on this site.", "limit": 0, "repeats": {"min": 2, "max": 6, "of": [{"name": "label", "usage": "The destination page name.", "limit": 60, "required": true}, {"name": "href", "usage": "A real path on this site.", "limit": 200, "required": true}, {"name": "body", "usage": "One sentence on what is there.", "limit": 160}]}}
|
|
69
|
+
{"name": "items", "usage": "Real related pages on this site.", "limit": 0, "repeats": {"min": 2, "max": 6, "of": [{"name": "label", "usage": "The destination page name.", "limit": 60, "required": true}, {"name": "href", "usage": "A real path on this site.", "limit": 200, "required": true, "translatable": false}, {"name": "body", "usage": "One sentence on what is there.", "limit": 160}]}}
|
|
70
70
|
],
|
|
71
71
|
"example": {"type": "links", "heading": "Keep exploring", "items": [{"label": "How it works", "href": "/how-it-works/", "body": "See the process from start to finish."}, {"label": "Frequently asked questions", "href": "/faq/", "body": "Find concise answers to common questions."}]}
|
|
72
72
|
},
|
|
@@ -24,6 +24,7 @@ from maggie_dash_store import MaggieDashStore # noqa: E402
|
|
|
24
24
|
from service_variants import ServiceVariantStore # noqa: E402
|
|
25
25
|
from maggie_dash_ui import load_and_validate # noqa: E402
|
|
26
26
|
from maggie_sections import catalogue, remap_translations, section_id_migration, validate_registry, copy_notes, validate_values, validate_fanout, reconcile_fields, validate_locale_coverage # noqa: E402
|
|
27
|
+
from maggie_quality import validate_variant_copy, validate_media_uniqueness, classify_inventory, validate_bindings, validate_reconcile_contract # noqa: E402
|
|
27
28
|
from route_imports import classify_bindings # noqa: E402
|
|
28
29
|
|
|
29
30
|
|
|
@@ -358,7 +359,29 @@ def command_sections(args: argparse.Namespace) -> int:
|
|
|
358
359
|
if args.sections_command == "locale-validate":
|
|
359
360
|
sections = json.loads(Path(args.sections_file).resolve().read_text(encoding="utf-8"))
|
|
360
361
|
translations = json.loads(Path(args.translations_file).resolve().read_text(encoding="utf-8"))
|
|
361
|
-
|
|
362
|
+
registry = json.loads(Path(args.registry).resolve().read_text(encoding="utf-8")) if args.registry else None
|
|
363
|
+
result = validate_locale_coverage(args.page_id, sections, translations, args.locale, registry=registry)
|
|
364
|
+
emit(result)
|
|
365
|
+
return 0 if result["passed"] else 1
|
|
366
|
+
if args.sections_command == "variant-copy-validate":
|
|
367
|
+
value = json.loads(Path(args.pages_file).resolve().read_text(encoding="utf-8"))
|
|
368
|
+
result = validate_variant_copy(value, args.field or ["faq"])
|
|
369
|
+
emit(result)
|
|
370
|
+
return 0 if result["passed"] else 1
|
|
371
|
+
if args.sections_command == "media-validate":
|
|
372
|
+
value = json.loads(Path(args.pages_file).resolve().read_text(encoding="utf-8"))
|
|
373
|
+
result = validate_media_uniqueness(value, across_siblings=args.across_siblings)
|
|
374
|
+
emit(result)
|
|
375
|
+
return 0 if result["passed"] else 1
|
|
376
|
+
if args.sections_command == "bindings-validate":
|
|
377
|
+
sections = json.loads(Path(args.sections_file).resolve().read_text(encoding="utf-8"))
|
|
378
|
+
references = json.loads(Path(args.references_file).resolve().read_text(encoding="utf-8"))
|
|
379
|
+
result = validate_bindings(sections, references)
|
|
380
|
+
emit(result)
|
|
381
|
+
return 0 if result["passed"] else 1
|
|
382
|
+
if args.sections_command == "idempotency-validate":
|
|
383
|
+
contract = json.loads(Path(args.contract_file).resolve().read_text(encoding="utf-8"))
|
|
384
|
+
result = validate_reconcile_contract(contract)
|
|
362
385
|
emit(result)
|
|
363
386
|
return 0 if result["passed"] else 1
|
|
364
387
|
registry = json.loads(Path(args.registry).resolve().read_text(encoding="utf-8"))
|
|
@@ -396,6 +419,14 @@ def command_components_audit(args: argparse.Namespace) -> int:
|
|
|
396
419
|
return 0 if result["passed"] else 1
|
|
397
420
|
|
|
398
421
|
|
|
422
|
+
def command_inventory(args: argparse.Namespace) -> int:
|
|
423
|
+
"""Classify each published page once, with no broad predicate overlap."""
|
|
424
|
+
value = json.loads(Path(args.pages_file).resolve().read_text(encoding="utf-8"))
|
|
425
|
+
result = classify_inventory(value)
|
|
426
|
+
emit(result)
|
|
427
|
+
return 0 if result["passed"] else 1
|
|
428
|
+
|
|
429
|
+
|
|
399
430
|
def variant_store(args: argparse.Namespace) -> ServiceVariantStore:
|
|
400
431
|
return ServiceVariantStore(project_root(args) / ".maggie" / "service-variants.json")
|
|
401
432
|
|
|
@@ -518,8 +549,20 @@ def parser() -> argparse.ArgumentParser:
|
|
|
518
549
|
sections_reconcile.add_argument("--current-file", required=True); sections_reconcile.add_argument("--desired-file", required=True); sections_reconcile.add_argument("--output", required=True); sections_reconcile.add_argument("--force", action="store_true"); sections_reconcile.add_argument("--confirm", action="store_true")
|
|
519
550
|
sections_reconcile.set_defaults(func=command_sections)
|
|
520
551
|
sections_locale = sections_sub.add_parser("locale-validate", help="check locale sidecar coverage for stored image alt fields")
|
|
521
|
-
sections_locale.add_argument("--page-id", required=True); sections_locale.add_argument("--sections-file", required=True); sections_locale.add_argument("--translations-file", required=True); sections_locale.add_argument("--locale", action="append", required=True)
|
|
552
|
+
sections_locale.add_argument("--page-id", required=True); sections_locale.add_argument("--sections-file", required=True); sections_locale.add_argument("--translations-file", required=True); sections_locale.add_argument("--locale", action="append", required=True); sections_locale.add_argument("--registry", help="registry for all declared translatable fields")
|
|
522
553
|
sections_locale.set_defaults(func=command_sections)
|
|
554
|
+
sections_variants = sections_sub.add_parser("variant-copy-validate", help="detect duplicated copy across sibling variants")
|
|
555
|
+
sections_variants.add_argument("--pages-file", required=True); sections_variants.add_argument("--field", action="append", default=[])
|
|
556
|
+
sections_variants.set_defaults(func=command_sections)
|
|
557
|
+
sections_media = sections_sub.add_parser("media-validate", help="detect duplicate media within pages or sibling families")
|
|
558
|
+
sections_media.add_argument("--pages-file", required=True); sections_media.add_argument("--across-siblings", action="store_true")
|
|
559
|
+
sections_media.set_defaults(func=command_sections)
|
|
560
|
+
sections_bindings = sections_sub.add_parser("bindings-validate", help="resolve stored section references against a reference inventory")
|
|
561
|
+
sections_bindings.add_argument("--sections-file", required=True); sections_bindings.add_argument("--references-file", required=True)
|
|
562
|
+
sections_bindings.set_defaults(func=command_sections)
|
|
563
|
+
sections_idempotency = sections_sub.add_parser("idempotency-validate", help="validate a migration repair selection and convergence contract")
|
|
564
|
+
sections_idempotency.add_argument("--contract-file", required=True)
|
|
565
|
+
sections_idempotency.set_defaults(func=command_sections)
|
|
523
566
|
sections_keys.set_defaults(func=command_sections)
|
|
524
567
|
components = sub.add_parser("components-audit", help="classify route component bindings against live section/page inventories")
|
|
525
568
|
components.add_argument("--bindings-file", required=True)
|
|
@@ -527,6 +570,9 @@ def parser() -> argparse.ArgumentParser:
|
|
|
527
570
|
components.add_argument("--pages-file", required=True, help="JSON array/object of live page paths")
|
|
528
571
|
components.add_argument("--output")
|
|
529
572
|
components.set_defaults(func=command_components_audit)
|
|
573
|
+
inventory = sub.add_parser("inventory", help="classify published pages into disjoint band/code-rendered kinds")
|
|
574
|
+
inventory.add_argument("--pages-file", required=True, help="JSON page inventory")
|
|
575
|
+
inventory.set_defaults(func=command_inventory)
|
|
530
576
|
variant = sub.add_parser("variant", help="manage service variant lifecycle")
|
|
531
577
|
variant_sub = variant.add_subparsers(dest="variant_command", required=True)
|
|
532
578
|
create = variant_sub.add_parser("create"); create.add_argument("--project", default="."); create.add_argument("--service-id", required=True); create.add_argument("--variant-id", required=True); create.add_argument("--variant-type", required=True); create.add_argument("--locale", required=True); create.add_argument("--market", required=True); create.add_argument("--slug", required=True); create.add_argument("--title", required=True); create.add_argument("--facts", required=True); create.add_argument("--source-revision", required=True); create.add_argument("--canonical-variant-id"); create.add_argument("--cluster-link", action="append", default=[]); create.add_argument("--layout-family", default="service-default"); create.add_argument("--confirm", action="store_true")
|
|
@@ -4,6 +4,7 @@
|
|
|
4
4
|
from __future__ import annotations
|
|
5
5
|
|
|
6
6
|
import argparse
|
|
7
|
+
import hashlib
|
|
7
8
|
import json
|
|
8
9
|
import re
|
|
9
10
|
import sys
|
|
@@ -54,7 +55,73 @@ def validate_project(project: Path, environment: str) -> dict:
|
|
|
54
55
|
return result
|
|
55
56
|
|
|
56
57
|
|
|
58
|
+
IDENTITY_SCHEMA = "maggie-database-identity.v1"
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _identity_fingerprint(identity: dict) -> str:
|
|
62
|
+
material = {key: identity.get(key) for key in ("environment", "targetName", "databaseName", "serverIdentity", "serviceIdentity", "rowCounts")}
|
|
63
|
+
return hashlib.sha256(json.dumps(material, sort_keys=True, separators=(",", ":"), ensure_ascii=False).encode()).hexdigest()
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def validate_identity(identity: object, expected: object | None = None) -> dict:
|
|
67
|
+
"""Gate a write against a non-secret database identity report.
|
|
68
|
+
|
|
69
|
+
The report intentionally contains names, server identity and row counts,
|
|
70
|
+
never a connection string. Comparing it with the service/deployment
|
|
71
|
+
identity catches the common case where staging and production credentials
|
|
72
|
+
resolve to different or unexpectedly identical databases.
|
|
73
|
+
"""
|
|
74
|
+
errors: list[str] = []
|
|
75
|
+
if not isinstance(identity, dict):
|
|
76
|
+
return {"schemaVersion": IDENTITY_SCHEMA, "passed": False, "errors": ["identity report must be an object"], "mutation": "not executed"}
|
|
77
|
+
if identity.get("schemaVersion") != IDENTITY_SCHEMA:
|
|
78
|
+
errors.append(f"schemaVersion must be {IDENTITY_SCHEMA}")
|
|
79
|
+
for field in ("environment", "targetName", "databaseName", "serverIdentity", "serviceIdentity"):
|
|
80
|
+
if not str(identity.get(field) or "").strip():
|
|
81
|
+
errors.append(f"{field} is required")
|
|
82
|
+
if not isinstance(identity.get("rowCounts"), dict):
|
|
83
|
+
errors.append("rowCounts must be an object")
|
|
84
|
+
fingerprint = str(identity.get("fingerprint") or "")
|
|
85
|
+
if not re.fullmatch(r"[0-9a-f]{64}", fingerprint):
|
|
86
|
+
errors.append("fingerprint must be a SHA-256 hex digest")
|
|
87
|
+
elif fingerprint != _identity_fingerprint(identity):
|
|
88
|
+
errors.append("fingerprint does not match the reported identity")
|
|
89
|
+
compared: list[str] = []
|
|
90
|
+
if expected is not None:
|
|
91
|
+
if not isinstance(expected, dict):
|
|
92
|
+
errors.append("expected identity report must be an object")
|
|
93
|
+
else:
|
|
94
|
+
for field in ("environment", "targetName", "databaseName", "serverIdentity", "serviceIdentity"):
|
|
95
|
+
if str(identity.get(field)) != str(expected.get(field)):
|
|
96
|
+
errors.append(f"target identity mismatch: {field}")
|
|
97
|
+
else:
|
|
98
|
+
compared.append(field)
|
|
99
|
+
return {"schemaVersion": IDENTITY_SCHEMA, "passed": not errors, "errors": errors, "mutation": "not executed", "compared": compared, "target": {key: identity.get(key) for key in ("environment", "targetName", "databaseName", "serverIdentity", "serviceIdentity", "rowCounts")}}
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def identity_main(argv: list[str]) -> int:
|
|
103
|
+
parser = argparse.ArgumentParser(description="Validate a secret-free database target identity before migration writes.")
|
|
104
|
+
parser.add_argument("--identity-file", required=True)
|
|
105
|
+
parser.add_argument("--expected-file")
|
|
106
|
+
parser.add_argument("--output")
|
|
107
|
+
args = parser.parse_args(argv)
|
|
108
|
+
try:
|
|
109
|
+
actual = json.loads(Path(args.identity_file).read_text(encoding="utf-8"))
|
|
110
|
+
expected = json.loads(Path(args.expected_file).read_text(encoding="utf-8")) if args.expected_file else None
|
|
111
|
+
result = validate_identity(actual, expected)
|
|
112
|
+
except (OSError, json.JSONDecodeError) as error:
|
|
113
|
+
result = {"schemaVersion": IDENTITY_SCHEMA, "passed": False, "errors": [f"cannot read identity report: {error}"], "mutation": "not executed"}
|
|
114
|
+
if args.output:
|
|
115
|
+
output = Path(args.output)
|
|
116
|
+
output.parent.mkdir(parents=True, exist_ok=True)
|
|
117
|
+
output.write_text(json.dumps(result, indent=2, ensure_ascii=False) + "\n", encoding="utf-8")
|
|
118
|
+
print(json.dumps(result, indent=2, ensure_ascii=False))
|
|
119
|
+
return 0 if result["passed"] else 1
|
|
120
|
+
|
|
121
|
+
|
|
57
122
|
def main() -> int:
|
|
123
|
+
if sys.argv[1:2] == ["identity"]:
|
|
124
|
+
return identity_main(sys.argv[2:])
|
|
58
125
|
parser = argparse.ArgumentParser(description=__doc__)
|
|
59
126
|
parser.add_argument("manifest", nargs="?")
|
|
60
127
|
parser.add_argument("--project")
|
|
@@ -257,6 +257,15 @@ def sitemap_urls(base: str) -> tuple[list[str], list[dict[str, str]]]:
|
|
|
257
257
|
return list(dict.fromkeys(pages)), violations
|
|
258
258
|
|
|
259
259
|
|
|
260
|
+
def access_log_sitemap_check(path: Path, sitemap_path: str = "/sitemap.xml") -> dict[str, object]:
|
|
261
|
+
"""Distinguish a sitemap fetch failure from a crawler that never asked."""
|
|
262
|
+
text = path.read_text(encoding="utf-8", errors="replace")
|
|
263
|
+
escaped = re.escape(sitemap_path)
|
|
264
|
+
requests = [line for line in text.splitlines() if re.search(rf"(?:GET|HEAD)\s+{escaped}(?:[?\s\"]|$)", line, re.I)]
|
|
265
|
+
googlebot = [line for line in requests if "googlebot" in line.lower()]
|
|
266
|
+
return {"provided": True, "path": str(path), "requestCount": len(requests), "googlebotRequestCount": len(googlebot), "status": "requested" if requests else "not-requested", "diagnosis": "googlebot-request-observed" if googlebot else ("other-client-request-only" if requests else "no-matching-request")}
|
|
267
|
+
|
|
268
|
+
|
|
260
269
|
def main() -> int:
|
|
261
270
|
parser = argparse.ArgumentParser()
|
|
262
271
|
parser.add_argument("url")
|
|
@@ -269,6 +278,8 @@ def main() -> int:
|
|
|
269
278
|
parser.add_argument("--markets", help="comma-separated markets: global,uk,us")
|
|
270
279
|
parser.add_argument("--check-hreflang", action="store_true")
|
|
271
280
|
parser.add_argument("--check-translation-completeness", action="store_true")
|
|
281
|
+
parser.add_argument("--access-log", type=Path, help="optional local access log for crawler-request diagnosis")
|
|
282
|
+
parser.add_argument("--require-sitemap-request", action="store_true", help="fail unless the supplied access log contains a sitemap request")
|
|
272
283
|
baseline_args = parser.add_mutually_exclusive_group()
|
|
273
284
|
baseline_args.add_argument("--save-baseline", type=Path, help="create a new reviewed contract from a passing complete crawl")
|
|
274
285
|
baseline_args.add_argument("--baseline", type=Path, help="fail on differences from a reviewed contract")
|
|
@@ -326,6 +337,12 @@ def main() -> int:
|
|
|
326
337
|
"mentions_sitemap": "sitemap" in body.lower() if name == "robots" else None,
|
|
327
338
|
"url_count": len(re.findall(r"<loc>.*?</loc>", body, re.I | re.S)) if name == "sitemap" else None,
|
|
328
339
|
}
|
|
340
|
+
if name == "sitemap" and args.access_log:
|
|
341
|
+
access = access_log_sitemap_check(args.access_log)
|
|
342
|
+
checks["sitemap_access_log"] = {"ok": not args.require_sitemap_request or access["requestCount"] > 0, **access}
|
|
343
|
+
checks[name]["access_log"] = access
|
|
344
|
+
if args.require_sitemap_request and not access["requestCount"]:
|
|
345
|
+
checks[name]["ok"] = False
|
|
329
346
|
except Exception as exc:
|
|
330
347
|
checks[name] = {"ok": False, "error": type(exc).__name__}
|
|
331
348
|
|
|
@@ -0,0 +1,231 @@
|
|
|
1
|
+
"""Provider-neutral quality contracts for variants, media, inventory, and bindings."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import hashlib
|
|
6
|
+
import json
|
|
7
|
+
import re
|
|
8
|
+
from collections import Counter, defaultdict
|
|
9
|
+
from typing import Any, Iterable
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
VARIANT_SCHEMA = "maggie-variant-copy.v1"
|
|
13
|
+
MEDIA_SCHEMA = "maggie-media-uniqueness.v1"
|
|
14
|
+
INVENTORY_SCHEMA = "maggie-page-inventory.v1"
|
|
15
|
+
BINDING_SCHEMA = "maggie-section-bindings.v1"
|
|
16
|
+
IDEMPOTENCY_SCHEMA = "maggie-reconcile-contract.v1"
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def _items(value: object, *keys: str) -> list[dict[str, Any]]:
|
|
20
|
+
if isinstance(value, list):
|
|
21
|
+
return [item for item in value if isinstance(item, dict)]
|
|
22
|
+
if isinstance(value, dict):
|
|
23
|
+
for key in keys:
|
|
24
|
+
candidate = value.get(key)
|
|
25
|
+
if isinstance(candidate, list):
|
|
26
|
+
return [item for item in candidate if isinstance(item, dict)]
|
|
27
|
+
return []
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def _normalise(value: object) -> object:
|
|
31
|
+
if isinstance(value, dict):
|
|
32
|
+
return {str(key): _normalise(value[key]) for key in sorted(value)}
|
|
33
|
+
if isinstance(value, list):
|
|
34
|
+
values = [_normalise(item) for item in value]
|
|
35
|
+
return sorted(values, key=lambda item: json.dumps(item, sort_keys=True, ensure_ascii=False))
|
|
36
|
+
if isinstance(value, str):
|
|
37
|
+
return re.sub(r"\s+", " ", value).strip().casefold()
|
|
38
|
+
return value
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _field_value(page: dict[str, Any], field: str) -> object:
|
|
42
|
+
if field in page:
|
|
43
|
+
return page[field]
|
|
44
|
+
for section in page.get("sections", []) if isinstance(page.get("sections"), list) else []:
|
|
45
|
+
if isinstance(section, dict) and (section.get("type") == field or section.get("id") == field):
|
|
46
|
+
copy = {key: value for key, value in section.items() if key not in {"id", "type"}}
|
|
47
|
+
return copy
|
|
48
|
+
return None
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def validate_variant_copy(value: object, fields: Iterable[str] = ("faq",)) -> dict[str, Any]:
|
|
52
|
+
pages = _items(value, "pages", "variants")
|
|
53
|
+
errors: list[str] = []
|
|
54
|
+
duplicate_groups: list[dict[str, Any]] = []
|
|
55
|
+
grouped: dict[tuple[str, str, str], list[dict[str, Any]]] = defaultdict(list)
|
|
56
|
+
field_list = [str(field) for field in fields if str(field)]
|
|
57
|
+
for index, page in enumerate(pages):
|
|
58
|
+
page_id = str(page.get("id") or page.get("variantId") or f"item-{index}")
|
|
59
|
+
family = str(page.get("family") or page.get("variantFamily") or page.get("sourceVariantKey") or page.get("canonicalVariantId") or "")
|
|
60
|
+
locale = str(page.get("locale") or "default")
|
|
61
|
+
if not family:
|
|
62
|
+
errors.append(f"{page_id}: variant family is required")
|
|
63
|
+
continue
|
|
64
|
+
for field in field_list:
|
|
65
|
+
selected = _field_value(page, field)
|
|
66
|
+
if selected is not None:
|
|
67
|
+
digest = hashlib.sha256(json.dumps(_normalise(selected), sort_keys=True, ensure_ascii=False).encode()).hexdigest()
|
|
68
|
+
grouped[(family, locale, field)].append({"id": page_id, "digest": digest})
|
|
69
|
+
for (family, locale, field), records in sorted(grouped.items()):
|
|
70
|
+
by_digest: dict[str, list[str]] = defaultdict(list)
|
|
71
|
+
for record in records:
|
|
72
|
+
by_digest[record["digest"]].append(record["id"])
|
|
73
|
+
for digest, ids in sorted(by_digest.items()):
|
|
74
|
+
if len(ids) > 1:
|
|
75
|
+
duplicate_groups.append({"family": family, "locale": locale, "field": field, "variantIds": sorted(ids), "digest": digest})
|
|
76
|
+
return {"schemaVersion": VARIANT_SCHEMA, "passed": not errors and not duplicate_groups, "errors": errors, "duplicates": duplicate_groups, "pages": len(pages), "fields": field_list}
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def _media_sources(value: object, path: str = "") -> list[dict[str, str]]:
|
|
80
|
+
media_keys = {"image", "imageurl", "image_url", "src", "srcset", "poster", "media", "mediaurl", "media_url"}
|
|
81
|
+
result: list[dict[str, str]] = []
|
|
82
|
+
if isinstance(value, dict):
|
|
83
|
+
for key, child in value.items():
|
|
84
|
+
child_path = f"{path}.{key}" if path else str(key)
|
|
85
|
+
if str(key).casefold() in media_keys and isinstance(child, str) and child.strip():
|
|
86
|
+
result.append({"source": child.strip(), "path": child_path})
|
|
87
|
+
else:
|
|
88
|
+
result.extend(_media_sources(child, child_path))
|
|
89
|
+
elif isinstance(value, list):
|
|
90
|
+
for index, child in enumerate(value):
|
|
91
|
+
result.extend(_media_sources(child, f"{path}.{index}"))
|
|
92
|
+
return result
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def validate_media_uniqueness(value: object, *, across_siblings: bool = False) -> dict[str, Any]:
|
|
96
|
+
pages = _items(value, "pages", "variants")
|
|
97
|
+
errors: list[dict[str, Any]] = []
|
|
98
|
+
cross: dict[tuple[str, str], list[str]] = defaultdict(list)
|
|
99
|
+
for index, page in enumerate(pages):
|
|
100
|
+
page_id = str(page.get("id") or page.get("variantId") or f"item-{index}")
|
|
101
|
+
sources = _media_sources(page.get("sections", page))
|
|
102
|
+
by_source: dict[str, list[str]] = defaultdict(list)
|
|
103
|
+
for item in sources:
|
|
104
|
+
by_source[item["source"]].append(item["path"])
|
|
105
|
+
for source, paths in sorted(by_source.items()):
|
|
106
|
+
if len(paths) > 1:
|
|
107
|
+
errors.append({"scope": "page", "pageId": page_id, "source": source, "paths": paths})
|
|
108
|
+
if across_siblings:
|
|
109
|
+
family = str(page.get("family") or page.get("variantFamily") or page.get("sourceVariantKey") or page_id)
|
|
110
|
+
for source in by_source:
|
|
111
|
+
cross[(family, source)].append(page_id)
|
|
112
|
+
if across_siblings:
|
|
113
|
+
for (family, source), ids in sorted(cross.items()):
|
|
114
|
+
if len(set(ids)) > 1:
|
|
115
|
+
errors.append({"scope": "siblings", "family": family, "source": source, "pageIds": sorted(set(ids))})
|
|
116
|
+
return {"schemaVersion": MEDIA_SCHEMA, "passed": not errors, "duplicates": errors, "pages": len(pages), "acrossSiblings": across_siblings}
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def classify_inventory(value: object) -> dict[str, Any]:
|
|
120
|
+
pages = _items(value, "pages", "items", "routes")
|
|
121
|
+
allowed = {"band", "code-rendered", "redirect", "asset", "unknown"}
|
|
122
|
+
records: list[dict[str, Any]] = []
|
|
123
|
+
errors: list[str] = []
|
|
124
|
+
seen: set[str] = set()
|
|
125
|
+
for index, page in enumerate(pages):
|
|
126
|
+
path = str(page.get("path") or page.get("url") or page.get("route") or f"item-{index}")
|
|
127
|
+
if path in seen:
|
|
128
|
+
errors.append(f"duplicate inventory path: {path}")
|
|
129
|
+
seen.add(path)
|
|
130
|
+
explicit = str(page.get("kind") or page.get("classification") or "").casefold()
|
|
131
|
+
if explicit in {"code", "code-rendered", "component", "source"}:
|
|
132
|
+
kind = "code-rendered"
|
|
133
|
+
elif explicit in {"band", "section", "sections"}:
|
|
134
|
+
kind = "band"
|
|
135
|
+
elif explicit in {"redirect", "asset"}:
|
|
136
|
+
kind = explicit
|
|
137
|
+
elif isinstance(page.get("sections"), list) or isinstance(page.get("sectionTypes"), list):
|
|
138
|
+
kind = "band"
|
|
139
|
+
elif page.get("renderedBy") or re.search(r"\.(astro|tsx|jsx|vue|svelte)$", str(page.get("source") or page.get("file") or path), re.I):
|
|
140
|
+
kind = "code-rendered"
|
|
141
|
+
else:
|
|
142
|
+
kind = "unknown"
|
|
143
|
+
records.append({"path": path, "kind": kind})
|
|
144
|
+
if kind not in allowed or kind == "unknown":
|
|
145
|
+
errors.append(f"{path}: cannot classify published page")
|
|
146
|
+
counts = dict(sorted(Counter(item["kind"] for item in records).items()))
|
|
147
|
+
return {"schemaVersion": INVENTORY_SCHEMA, "passed": not errors and len(records) == len(seen), "records": records, "counts": counts, "bandCount": counts.get("band", 0), "codeRenderedCount": counts.get("code-rendered", 0), "errors": errors, "disjoint": True}
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def _reference_values(references: object) -> dict[str, set[str]]:
|
|
151
|
+
source = references if isinstance(references, dict) else {}
|
|
152
|
+
result: dict[str, set[str]] = {}
|
|
153
|
+
for key, values in source.items():
|
|
154
|
+
entries = values if isinstance(values, list) else [values]
|
|
155
|
+
allowed: set[str] = set()
|
|
156
|
+
for item in entries:
|
|
157
|
+
if isinstance(item, dict):
|
|
158
|
+
for candidate in (item.get("id"), item.get("slug"), item.get("name"), item.get("title"), item.get("value")):
|
|
159
|
+
if candidate is not None and str(candidate).strip():
|
|
160
|
+
allowed.add(str(candidate).strip())
|
|
161
|
+
elif isinstance(item, (str, int)):
|
|
162
|
+
allowed.add(str(item).strip())
|
|
163
|
+
result[str(key).casefold()] = allowed
|
|
164
|
+
return result
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def validate_bindings(sections: object, references: object) -> dict[str, Any]:
|
|
168
|
+
allowed = _reference_values(references)
|
|
169
|
+
unresolved: list[dict[str, str]] = []
|
|
170
|
+
checked: list[dict[str, str]] = []
|
|
171
|
+
suffixes = ("id", "slug", "ref", "reference")
|
|
172
|
+
values = sections if isinstance(sections, list) else sections.get("sections", []) if isinstance(sections, dict) else []
|
|
173
|
+
for index, section in enumerate(values):
|
|
174
|
+
if not isinstance(section, dict):
|
|
175
|
+
continue
|
|
176
|
+
section_id = str(section.get("id") or index)
|
|
177
|
+
for key, value in section.items():
|
|
178
|
+
if not isinstance(value, (str, int)) or not str(value).strip():
|
|
179
|
+
continue
|
|
180
|
+
key_lower = str(key).casefold()
|
|
181
|
+
base = key_lower
|
|
182
|
+
for suffix in suffixes:
|
|
183
|
+
if base.endswith(suffix) and len(base) > len(suffix):
|
|
184
|
+
base = base[: -len(suffix)]
|
|
185
|
+
break
|
|
186
|
+
if base not in allowed:
|
|
187
|
+
continue
|
|
188
|
+
record = {"sectionId": section_id, "field": str(key), "value": str(value)}
|
|
189
|
+
checked.append(record)
|
|
190
|
+
if str(value) not in allowed[base]:
|
|
191
|
+
unresolved.append(record)
|
|
192
|
+
binding = section.get("binding")
|
|
193
|
+
if isinstance(binding, dict):
|
|
194
|
+
binding_type = str(binding.get("type") or "").casefold()
|
|
195
|
+
binding_value = binding.get("value", binding.get("slug", binding.get("id", binding.get("ref"))))
|
|
196
|
+
if binding_type and binding_value is not None and binding_type in allowed:
|
|
197
|
+
record = {"sectionId": section_id, "field": "binding", "value": str(binding_value)}
|
|
198
|
+
checked.append(record)
|
|
199
|
+
if str(binding_value) not in allowed[binding_type]:
|
|
200
|
+
unresolved.append(record)
|
|
201
|
+
return {"schemaVersion": BINDING_SCHEMA, "passed": not unresolved, "checked": checked, "unresolved": unresolved, "referenceTypes": sorted(allowed)}
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def validate_reconcile_contract(contract: object) -> dict[str, Any]:
|
|
205
|
+
errors: list[str] = []
|
|
206
|
+
if not isinstance(contract, dict):
|
|
207
|
+
return {"schemaVersion": IDEMPOTENCY_SCHEMA, "passed": False, "errors": ["reconcile contract must be an object"]}
|
|
208
|
+
if contract.get("schemaVersion") not in {None, IDEMPOTENCY_SCHEMA}:
|
|
209
|
+
errors.append(f"schemaVersion must be {IDEMPOTENCY_SCHEMA}")
|
|
210
|
+
candidate = contract.get("candidateSelection") if isinstance(contract.get("candidateSelection"), dict) else {}
|
|
211
|
+
strategy = str(candidate.get("strategy") or "")
|
|
212
|
+
if strategy not in {"all-candidates", "all_rows", "all-rows"}:
|
|
213
|
+
errors.append("candidateSelection.strategy must select all candidates")
|
|
214
|
+
if candidate.get("filtersOnlyUnconverted") is True:
|
|
215
|
+
errors.append("candidate selection cannot filter only rows that look converted")
|
|
216
|
+
query = str(candidate.get("query") or "").casefold()
|
|
217
|
+
if query and re.search(r"(is\s+null|=\s*'?(empty|converted|none)|jsonb_array_length\s*\([^)]*\)\s*=\s*0)", query):
|
|
218
|
+
errors.append("candidate selection query looks limited to already-unconverted rows")
|
|
219
|
+
compare_fields = contract.get("compareFields")
|
|
220
|
+
if not isinstance(compare_fields, list) or not compare_fields:
|
|
221
|
+
errors.append("compareFields must list every repaired field")
|
|
222
|
+
if contract.get("writesOnlyWhenChanged") is not True:
|
|
223
|
+
errors.append("writesOnlyWhenChanged must be true")
|
|
224
|
+
report = contract.get("report") if isinstance(contract.get("report"), dict) else {}
|
|
225
|
+
for key in ("changed", "unchanged"):
|
|
226
|
+
if report.get(key) is not True:
|
|
227
|
+
errors.append(f"report.{key} evidence is required")
|
|
228
|
+
second = contract.get("secondRun") if isinstance(contract.get("secondRun"), dict) else {}
|
|
229
|
+
if second.get("convergent") is not True:
|
|
230
|
+
errors.append("secondRun.convergent must be true")
|
|
231
|
+
return {"schemaVersion": IDEMPOTENCY_SCHEMA, "passed": not errors, "errors": errors, "mutation": "not executed", "contract": {"candidateStrategy": strategy, "compareFields": compare_fields if isinstance(compare_fields, list) else [], "writesOnlyWhenChanged": contract.get("writesOnlyWhenChanged"), "report": report, "secondRun": second}}
|
|
@@ -193,31 +193,79 @@ def validate_values(registry: dict[str, Any], sections: object) -> dict[str, Any
|
|
|
193
193
|
return {"passed": not errors, "errors": errors}
|
|
194
194
|
|
|
195
195
|
|
|
196
|
-
def
|
|
197
|
-
|
|
196
|
+
def _registry_section_fields(registry: dict[str, Any] | None, section_type: str) -> list[dict[str, Any]]:
|
|
197
|
+
if not isinstance(registry, dict):
|
|
198
|
+
return []
|
|
199
|
+
for section in registry.get("sections", []):
|
|
200
|
+
if isinstance(section, dict) and str(section.get("type")) == section_type:
|
|
201
|
+
return [field for field in section.get("fields", []) if isinstance(field, dict)]
|
|
202
|
+
return []
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def _field_translatable(field: dict[str, Any]) -> bool:
|
|
206
|
+
"""Return the explicit copy policy, defaulting to translatable for text."""
|
|
207
|
+
return field.get("translatable") is not False
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def translation_key_paths(page_id: str, sections: object, field_names: tuple[str, ...] | None = ("imageAlt",), registry: dict[str, Any] | None = None) -> list[str]:
|
|
211
|
+
"""Find stored section fields that need a locale sidecar value.
|
|
212
|
+
|
|
213
|
+
Older callers can continue to pass ``field_names`` and receive the legacy
|
|
214
|
+
image-alt contract. New callers pass a registry, which makes every field
|
|
215
|
+
explicitly marked ``translatable`` part of the locale contract, including
|
|
216
|
+
repeated item fields.
|
|
217
|
+
"""
|
|
198
218
|
if not isinstance(sections, list):
|
|
199
219
|
return []
|
|
200
220
|
paths: list[str] = []
|
|
201
221
|
|
|
202
|
-
def
|
|
222
|
+
def walk_legacy(value: object, path: str) -> None:
|
|
203
223
|
if isinstance(value, dict):
|
|
204
224
|
for key, child in value.items():
|
|
205
225
|
child_path = f"{path}.{key}"
|
|
206
|
-
if key in field_names and isinstance(child, str) and child.strip():
|
|
226
|
+
if field_names and key in field_names and isinstance(child, str) and child.strip():
|
|
207
227
|
paths.append(child_path)
|
|
208
|
-
|
|
228
|
+
walk_legacy(child, child_path)
|
|
209
229
|
elif isinstance(value, list):
|
|
210
230
|
for index, child in enumerate(value):
|
|
211
|
-
|
|
231
|
+
walk_legacy(child, f"{path}.{index}")
|
|
232
|
+
|
|
233
|
+
def walk_declared(value: object, fields: list[dict[str, Any]], path: str) -> None:
|
|
234
|
+
if not isinstance(value, dict):
|
|
235
|
+
return
|
|
236
|
+
for field in fields:
|
|
237
|
+
name = str(field.get("name") or "")
|
|
238
|
+
if not name:
|
|
239
|
+
continue
|
|
240
|
+
child = value.get(name)
|
|
241
|
+
child_path = f"{path}.{name}"
|
|
242
|
+
if _field_translatable(field) and isinstance(child, str) and child.strip():
|
|
243
|
+
paths.append(child_path)
|
|
244
|
+
repeats = field.get("repeats")
|
|
245
|
+
if not isinstance(repeats, dict) or not isinstance(child, list):
|
|
246
|
+
continue
|
|
247
|
+
nested = [item for item in repeats.get("of", []) if isinstance(item, dict)]
|
|
248
|
+
for index, item in enumerate(child):
|
|
249
|
+
item_path = f"{child_path}.{index}"
|
|
250
|
+
if isinstance(item, dict):
|
|
251
|
+
walk_declared(item, nested, item_path)
|
|
252
|
+
elif len(nested) == 1 and _field_translatable(nested[0]) and isinstance(item, str) and item.strip():
|
|
253
|
+
paths.append(item_path)
|
|
212
254
|
|
|
213
255
|
for index, section in enumerate(sections):
|
|
214
|
-
|
|
256
|
+
path = f"page.{page_id}.sections.{section_key(section, index)}"
|
|
257
|
+
if registry and isinstance(section, dict):
|
|
258
|
+
declared = _registry_section_fields(registry, str(section.get("type") or ""))
|
|
259
|
+
if declared:
|
|
260
|
+
walk_declared(section, declared, path)
|
|
261
|
+
continue
|
|
262
|
+
walk_legacy(section, path)
|
|
215
263
|
return paths
|
|
216
264
|
|
|
217
265
|
|
|
218
|
-
def validate_locale_coverage(page_id: str, sections: object, translations: object, locales: Iterable[str]) -> dict[str, Any]:
|
|
266
|
+
def validate_locale_coverage(page_id: str, sections: object, translations: object, locales: Iterable[str], registry: dict[str, Any] | None = None) -> dict[str, Any]:
|
|
219
267
|
"""Require non-default locale rows for every declared translatable field."""
|
|
220
|
-
paths = translation_key_paths(page_id, sections)
|
|
268
|
+
paths = translation_key_paths(page_id, sections, registry=registry)
|
|
221
269
|
values = translations if isinstance(translations, dict) else {}
|
|
222
270
|
missing = [{"locale": locale, "key": key} for locale in locales for key in paths if not isinstance(values.get(locale), dict) or not str(values[locale].get(key) or "").strip()]
|
|
223
271
|
return {"schemaVersion": LOCALE_SCHEMA, "passed": not missing, "pageId": page_id, "requiredKeys": paths, "missing": missing, "writePolicy": "section row and locale rows must be committed in one transaction"}
|
|
@@ -309,12 +357,64 @@ def validate_fanout(registry: dict[str, Any], fanout: object) -> dict[str, Any]:
|
|
|
309
357
|
types = {str(item.get("type")) for item in registry.get("sections", []) if isinstance(item, dict)}
|
|
310
358
|
surfaces = ("typeUnion", "sectionSchema", "blank", "validation", "textExtraction", "renderer", "editor")
|
|
311
359
|
missing: list[dict[str, str]] = []
|
|
360
|
+
|
|
361
|
+
def field_paths(section: dict[str, Any]) -> set[str]:
|
|
362
|
+
result: set[str] = set()
|
|
363
|
+
for field in section.get("fields", []):
|
|
364
|
+
if not isinstance(field, dict) or not field.get("name"):
|
|
365
|
+
continue
|
|
366
|
+
name = str(field["name"])
|
|
367
|
+
result.add(name)
|
|
368
|
+
repeats = field.get("repeats")
|
|
369
|
+
if isinstance(repeats, dict):
|
|
370
|
+
for nested in repeats.get("of", []):
|
|
371
|
+
if isinstance(nested, dict) and nested.get("name"):
|
|
372
|
+
result.add(f"{name}.{nested['name']}")
|
|
373
|
+
return result
|
|
374
|
+
|
|
375
|
+
required_fields = {
|
|
376
|
+
str(section.get("type")): field_paths(section)
|
|
377
|
+
for section in registry.get("sections", [])
|
|
378
|
+
if isinstance(section, dict) and section.get("type")
|
|
379
|
+
}
|
|
380
|
+
|
|
381
|
+
def declared(values: object) -> tuple[set[str], dict[str, set[str]]]:
|
|
382
|
+
if isinstance(values, list):
|
|
383
|
+
return set(map(str, values)), {}
|
|
384
|
+
if not isinstance(values, dict):
|
|
385
|
+
return set(), {}
|
|
386
|
+
types = set(map(str, values.keys()))
|
|
387
|
+
fields: dict[str, set[str]] = {}
|
|
388
|
+
for section_type, entry in values.items():
|
|
389
|
+
if isinstance(entry, list):
|
|
390
|
+
fields[str(section_type)] = set(map(str, entry))
|
|
391
|
+
elif isinstance(entry, dict):
|
|
392
|
+
candidate = entry.get("fields", [])
|
|
393
|
+
if isinstance(candidate, list):
|
|
394
|
+
fields[str(section_type)] = set(map(str, candidate))
|
|
395
|
+
return types, fields
|
|
396
|
+
|
|
312
397
|
for surface in surfaces:
|
|
313
398
|
values = fanout.get(surface)
|
|
314
|
-
|
|
315
|
-
for section_type in sorted(types -
|
|
399
|
+
declared_types, declared_fields = declared(values)
|
|
400
|
+
for section_type in sorted(types - declared_types):
|
|
316
401
|
missing.append({"type": section_type, "surface": surface})
|
|
317
|
-
|
|
402
|
+
for section_type in sorted(types & declared_types):
|
|
403
|
+
for field in sorted(required_fields.get(section_type, set()) - declared_fields.get(section_type, set())):
|
|
404
|
+
missing.append({"type": section_type, "surface": f"{surface}.{field}"})
|
|
405
|
+
|
|
406
|
+
editor_screens = fanout.get("editorScreens")
|
|
407
|
+
if not isinstance(editor_screens, dict) or len(editor_screens) < 2:
|
|
408
|
+
missing.append({"type": "editor", "surface": "editorScreens (at least two screens required)"})
|
|
409
|
+
elif isinstance(editor_screens, dict):
|
|
410
|
+
for screen, values in editor_screens.items():
|
|
411
|
+
screen_types, screen_fields = declared(values)
|
|
412
|
+
for section_type in sorted(types - screen_types):
|
|
413
|
+
missing.append({"type": section_type, "surface": f"editorScreens.{screen}"})
|
|
414
|
+
for section_type in sorted(types & screen_types):
|
|
415
|
+
for field in sorted(required_fields.get(section_type, set()) - screen_fields.get(section_type, set())):
|
|
416
|
+
missing.append({"type": section_type, "surface": f"editorScreens.{screen}.{field}"})
|
|
417
|
+
return {"schemaVersion": FANOUT_SCHEMA, "passed": not missing, "errors": [f"{item['type']} missing {item['surface']} fan-out" for item in missing], "sectionTypes": sorted(types), "missing": missing, "fieldAware": True, "editorScreenCount": len(editor_screens) if isinstance(editor_screens, dict) else 0}
|
|
318
418
|
|
|
319
419
|
|
|
320
420
|
def reconcile_fields(current: object, desired: object) -> dict[str, Any]:
|
|
@@ -14,6 +14,7 @@ HARD_URL_LIMIT = 50_000
|
|
|
14
14
|
HARD_BYTES_LIMIT = 52_428_800
|
|
15
15
|
DEFAULT_CHUNK_TARGET = 500
|
|
16
16
|
EXCLUDED_PATHS = re.compile(r"/(search|find|login|draft|preview)(/|$)", re.I)
|
|
17
|
+
W3C_UTC_DATETIME = re.compile(r"^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}Z$")
|
|
17
18
|
|
|
18
19
|
|
|
19
20
|
def absolute_url(url: str, origin: str) -> bool:
|
|
@@ -74,13 +75,29 @@ def filter_routes(routes: list[dict[str, str]], origin: str, content_types: set[
|
|
|
74
75
|
return groups, excluded
|
|
75
76
|
|
|
76
77
|
|
|
78
|
+
def conventional_lastmod(value: str) -> str:
|
|
79
|
+
"""Render date-only source evidence as a W3C/ISO UTC datetime."""
|
|
80
|
+
if re.fullmatch(r"\d{4}-\d{2}-\d{2}", value):
|
|
81
|
+
return value + "T00:00:00Z"
|
|
82
|
+
try:
|
|
83
|
+
parsed = datetime.fromisoformat(value.replace("Z", "+00:00"))
|
|
84
|
+
except ValueError:
|
|
85
|
+
return value
|
|
86
|
+
if parsed.tzinfo is None:
|
|
87
|
+
parsed = parsed.replace(tzinfo=timezone.utc)
|
|
88
|
+
return parsed.astimezone(timezone.utc).isoformat(timespec="seconds").replace("+00:00", "Z")
|
|
89
|
+
|
|
90
|
+
|
|
77
91
|
def xml_file(urls: list[dict[str, str]]) -> str:
|
|
78
92
|
rows = []
|
|
79
93
|
for item in urls:
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
94
|
+
fields = [f" <loc>{escape(item['url'])}</loc>"]
|
|
95
|
+
if item.get("lastmod"):
|
|
96
|
+
fields.append(f" <lastmod>{escape(conventional_lastmod(item['lastmod']))}</lastmod>")
|
|
97
|
+
fields.extend(f' <xhtml:link rel="alternate" hreflang="{escape(str(lang))}" href="{escape(str(url))}" />' for lang, url in sorted((item.get("alternates") or {}).items()))
|
|
98
|
+
rows.append(" <url>\n" + "\n".join(fields) + "\n </url>")
|
|
99
|
+
body = "\n".join(rows)
|
|
100
|
+
return '<?xml version="1.0" encoding="UTF-8"?>\n<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9" xmlns:xhtml="http://www.w3.org/1999/xhtml">\n' + body + ('\n' if body else '') + '</urlset>\n'
|
|
84
101
|
|
|
85
102
|
|
|
86
103
|
def build_plan(routes: list[dict[str, str]], origin: str, content_types: set[str], chunk_target: int = DEFAULT_CHUNK_TARGET, previous: dict | None = None) -> dict:
|
|
@@ -100,11 +117,12 @@ def build_plan(routes: list[dict[str, str]], origin: str, content_types: set[str
|
|
|
100
117
|
chunks.append({"filename": filename, "url": url, "entries": len(chunk_entries), "routes": chunk_entries, "bytes": len(xml.encode()), "sha256": hashlib.sha256(xml.encode()).hexdigest(), "xml": xml})
|
|
101
118
|
sitemap_urls.append(url)
|
|
102
119
|
group_plans.append({"contentType": content_type, "chunks": chunks})
|
|
103
|
-
|
|
120
|
+
index_rows = "\n".join(f" <sitemap>\n <loc>{escape(url)}</loc>\n </sitemap>" for url in sitemap_urls)
|
|
121
|
+
index_xml = '<?xml version="1.0" encoding="UTF-8"?>\n<sitemapindex xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">\n' + index_rows + ('\n' if index_rows else '') + '</sitemapindex>\n'
|
|
104
122
|
current = set(sitemap_urls)
|
|
105
123
|
old = set((previous or {}).get("sitemapUrls", []))
|
|
106
124
|
redirects = [{"from": url, "to": origin.rstrip("/") + "/sitemap.xml", "reason": "previously advertised sitemap removed"} for url in sorted(old - current)]
|
|
107
|
-
plan = {"schemaVersion": "maggie-seo-sitemap-plan.v1", "planId": "sitemap-" + datetime.now(timezone.utc).strftime("%Y%m%d%H%M%S"), "origin": origin.rstrip("/"), "groups": group_plans, "index": {"url": origin.rstrip("/") + "/sitemap.xml", "bytes": len(index_xml.encode()), "xml": index_xml, "sitemapUrls": sitemap_urls}, "redirects": redirects, "excluded": excluded}
|
|
125
|
+
plan = {"schemaVersion": "maggie-seo-sitemap-plan.v1", "planId": "sitemap-" + datetime.now(timezone.utc).strftime("%Y%m%d%H%M%S"), "origin": origin.rstrip("/"), "groups": group_plans, "index": {"url": origin.rstrip("/") + "/sitemap.xml", "bytes": len(index_xml.encode()), "xml": index_xml, "sitemapUrls": sitemap_urls, "responseHeaders": {"content-type": "text/xml; charset=utf-8", "x-robots-tag": "noindex"}}, "redirects": redirects, "excluded": excluded}
|
|
108
126
|
plan["validation"] = validate_plan_data(plan)
|
|
109
127
|
return plan
|
|
110
128
|
|
|
@@ -146,9 +164,16 @@ def validate_plan_data(plan: dict, strict_semantic: bool = False) -> dict:
|
|
|
146
164
|
for link in root.iter():
|
|
147
165
|
if link.tag.endswith("link") and link.get("rel") == "alternate" and not absolute_url(link.get("href", ""), origin):
|
|
148
166
|
errors.append(f"chunk contains a relative or off-origin alternate: {chunk.get('filename')}")
|
|
167
|
+
if link.tag.endswith("lastmod") and not W3C_UTC_DATETIME.fullmatch((link.text or "").strip()):
|
|
168
|
+
errors.append(f"lastmod must be a full W3C UTC datetime: {chunk.get('filename')}")
|
|
149
169
|
except ElementTree.ParseError:
|
|
150
170
|
errors.append(f"chunk is not valid XML: {chunk.get('filename')}")
|
|
151
171
|
index = plan.get("index", {})
|
|
172
|
+
headers = index.get("responseHeaders") if isinstance(index.get("responseHeaders"), dict) else {}
|
|
173
|
+
if headers.get("content-type") != "text/xml; charset=utf-8":
|
|
174
|
+
errors.append("sitemap response content-type must be text/xml; charset=utf-8")
|
|
175
|
+
if str(headers.get("x-robots-tag") or "").casefold() != "noindex":
|
|
176
|
+
errors.append("sitemap response x-robots-tag must be noindex")
|
|
152
177
|
if index.get("bytes", 0) > HARD_BYTES_LIMIT:
|
|
153
178
|
errors.append("sitemap index exceeds byte limit")
|
|
154
179
|
try:
|