@topy-ai/maggie 0.7.14 → 0.7.15
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +18 -4
- package/README.zh-TW.md +4 -1
- package/bin/maggie.js +6 -5
- package/bundled-skills/maggie-blog/SKILL.md +13 -9
- package/bundled-skills/maggie-dash/SKILL.md +17 -2
- package/bundled-skills/maggie-seo-geo/SKILL.md +21 -0
- package/bundled-tools/clis/maggie_blog.py +5 -1
- package/bundled-tools/clis/maggie_dash.py +14 -1
- package/bundled-tools/clis/maggie_indexnow.py +60 -0
- package/bundled-tools/runtime/maggie_api_contract.py +97 -0
- package/bundled-tools/runtime/maggie_blog.py +42 -1
- package/bundled-tools/runtime/maggie_indexnow.py +128 -0
- package/bundled-tools/runtime/maggie_quality.py +21 -4
- package/bundled-tools/runtime/maggie_sitemap.py +52 -1
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -76,6 +76,14 @@ maggie qa summary --project . --run <run-id>
|
|
|
76
76
|
The QA workflow stores secret-free run state under `.maggie/qa-runs/` and
|
|
77
77
|
keeps project-specific scenarios and evidence outside the npm package.
|
|
78
78
|
|
|
79
|
+
The current package also includes reusable safeguards from the latest feedback
|
|
80
|
+
review: `maggie dash api-contract` checks declared request and 2xx response
|
|
81
|
+
shapes; `maggie blog check-gate`/`approve` enforces review before publish;
|
|
82
|
+
sitemap validation flags suspiciously uniform `lastmod` dates; `maggie seo
|
|
83
|
+
indexnow` plans only changed same-origin URLs and holds back URLs accepted in
|
|
84
|
+
the last 24 hours; and dash inventory separates renderer kind from public page
|
|
85
|
+
kind while reporting source coverage for code-rendered routes.
|
|
86
|
+
|
|
79
87
|
For Google integrations, validate a redacted provider matrix before reporting
|
|
80
88
|
access. The command fails closed on unknown scopes, missing Ads prerequisites,
|
|
81
89
|
duplicate provider resources, and unverified edit/publish claims:
|
|
@@ -121,7 +129,8 @@ maggie dash install | init | status | migrate | cms ...
|
|
|
121
129
|
maggie dash transition ... # explicit content approval transition
|
|
122
130
|
maggie dash variant ... # service variant create/review/preview/publish
|
|
123
131
|
maggie dash sections ... # field fan-out, locale, binding, media, copy, identity keys
|
|
124
|
-
maggie dash inventory ... # disjoint
|
|
132
|
+
maggie dash inventory ... # disjoint renderer/page-kind inventory + coverage
|
|
133
|
+
maggie dash api-contract ... # declared API body and 2xx response contract
|
|
125
134
|
maggie agent-content write ... # host-authorized, origin-bound content bridge
|
|
126
135
|
maggie verification coverage ... # changed surface/locale evidence gate
|
|
127
136
|
maggie clone ... # authorized homepage capture
|
|
@@ -136,6 +145,7 @@ maggie service ... # import, sync, generate, validate
|
|
|
136
145
|
maggie seo performance ... # sampled PageSpeed/CWV report and baseline
|
|
137
146
|
maggie seo images ... # inventory, variants, confirmation, validate
|
|
138
147
|
maggie seo sitemap ... # typed/semantic plan, agent-files, apply, rollback
|
|
148
|
+
maggie seo indexnow ... # changed URLs, key check, retry-safe 24h guard
|
|
139
149
|
maggie deployment | migration | release | analytics | schedule
|
|
140
150
|
maggie migration identity --identity-file FILE [--expected-file FILE]
|
|
141
151
|
maggie deployment canary --asset URL=SHA256 --render-report report.json
|
|
@@ -237,8 +247,8 @@ artifact schemas.
|
|
|
237
247
|
Recommended upgrade sequence for the current release:
|
|
238
248
|
|
|
239
249
|
```bash
|
|
240
|
-
npx @topy-ai/maggie@0.7.
|
|
241
|
-
npx @topy-ai/maggie@0.7.
|
|
250
|
+
npx @topy-ai/maggie@0.7.15 update --project . --force
|
|
251
|
+
npx @topy-ai/maggie@0.7.15 cleanup --project .
|
|
242
252
|
```
|
|
243
253
|
|
|
244
254
|
Maintainers should pass npm credentials through the repository helper, never
|
|
@@ -248,7 +258,9 @@ as a command-line argument:
|
|
|
248
258
|
node scripts/publish-npm.mjs --maggie-env-file ../.env
|
|
249
259
|
```
|
|
250
260
|
|
|
251
|
-
The 0.7.
|
|
261
|
+
The 0.7.15 workflow adds API contracts, blog review gates, sitemap freshness,
|
|
262
|
+
change-driven IndexNow, and page-kind/source-coverage inventory. It also keeps
|
|
263
|
+
the general `maggie-qa-workflow` skill and `maggie qa`
|
|
252
264
|
CLI for scenario manifests, secret-free browser evidence metadata, test/fix/
|
|
253
265
|
retest lifecycle, adjacent regression checks, and explicit release gates. The
|
|
254
266
|
0.7.13 workflow adds field-aware section fan-out and locale coverage,
|
|
@@ -565,6 +577,8 @@ maggie design author --project . --route /about --purpose "Explain our approach"
|
|
|
565
577
|
maggie blog init --project . --confirm
|
|
566
578
|
maggie blog ingest --project . --source local --input content/posts.json --confirm
|
|
567
579
|
maggie blog validate --project .
|
|
580
|
+
maggie blog check-gate --project .
|
|
581
|
+
maggie blog approve --project . --slug example-post --actor reviewer --reason "reviewed" --confirm
|
|
568
582
|
maggie blog sitemap --project .
|
|
569
583
|
```
|
|
570
584
|
|
package/README.zh-TW.md
CHANGED
|
@@ -8,7 +8,7 @@ Codex、Claude Code 與相容的 coding agents。
|
|
|
8
8
|
## 安裝
|
|
9
9
|
|
|
10
10
|
```bash
|
|
11
|
-
npx @topy-ai/maggie@0.7.
|
|
11
|
+
npx @topy-ai/maggie@0.7.15 init --agent all
|
|
12
12
|
npx @topy-ai/maggie doctor --project .
|
|
13
13
|
```
|
|
14
14
|
|
|
@@ -25,6 +25,9 @@ maggie doctor --project . --require-bootstrap --strict
|
|
|
25
25
|
deployment、memory、feedback 和 MaggieDash。內容先 draft/review,外部寫入、
|
|
26
26
|
publish 與 production deployment 需要明確確認。
|
|
27
27
|
|
|
28
|
+
0.7.15 也加入 API schema contract、blog review gate、sitemap freshness
|
|
29
|
+
warning、change-driven IndexNow 與 page-kind/source-coverage inventory。
|
|
30
|
+
|
|
28
31
|
MaggieDash 也提供穩定 section identity、可重用 section arrangement、機器可讀
|
|
29
32
|
section registry、短期 agent content bridge,以及 translation out-of-band write
|
|
30
33
|
後的 restart gate。`maggie doctor` 會比較 install manifest 與磁碟上的實際 skills。
|
package/bin/maggie.js
CHANGED
|
@@ -59,7 +59,8 @@ Usage:
|
|
|
59
59
|
maggie bootstrap interview [project]
|
|
60
60
|
maggie dash init|install|status|migrate|transition|variant|cms --project PATH [options]
|
|
61
61
|
maggie dash sections <validate|prompt|keys|remap-translations|fanout-validate|locale-validate|variant-copy-validate|media-validate|bindings-validate|idempotency-validate|reconcile> [options]
|
|
62
|
-
maggie dash inventory --pages-file FILE
|
|
62
|
+
maggie dash inventory --pages-file FILE [--require-page-kinds] [--require-source-coverage]
|
|
63
|
+
maggie dash api-contract --spec FILE
|
|
63
64
|
maggie dash components-audit --bindings-file FILE --sections-file FILE --pages-file FILE
|
|
64
65
|
maggie dash status --project PATH
|
|
65
66
|
maggie dash migrate --project PATH --confirm
|
|
@@ -92,7 +93,7 @@ Usage:
|
|
|
92
93
|
maggie design step <job-id> --step NAME --evidence FILE
|
|
93
94
|
maggie auth reference --project PATH --confirm
|
|
94
95
|
maggie auth check --project PATH [--production]
|
|
95
|
-
maggie blog init|inspect|ingest|validate|publish|sitemap|settings|rollback|integration-state
|
|
96
|
+
maggie blog init|inspect|ingest|validate|check-gate|approve|publish|sitemap|settings|rollback|integration-state
|
|
96
97
|
maggie design status <job-id>
|
|
97
98
|
maggie service import <provider-url> --project PATH
|
|
98
99
|
maggie service sync <provider-url> --project PATH
|
|
@@ -112,7 +113,7 @@ Usage:
|
|
|
112
113
|
maggie api lifecycle --project PATH [--execute --allow-quota]
|
|
113
114
|
maggie memory <init|list|search|context|add|record-error|transition|export> --project PATH
|
|
114
115
|
maggie localization <extract|plan|generate|preview|validate|review|publish|stale|glossary> [options]
|
|
115
|
-
maggie seo performance|images|sitemap [options] (sitemap supports strict validate and agent-files)
|
|
116
|
+
maggie seo performance|images|sitemap|indexnow [options] (sitemap supports strict validate and agent-files)
|
|
116
117
|
maggie feedback <collect|preview|submit|list> [options]
|
|
117
118
|
maggie qa <start|record|summary|export> [options]
|
|
118
119
|
maggie site-audit URL [--crawl] [--access-log FILE] [--require-sitemap-request] [--languages en-GB,es-MX,ja-JP] [--check-hreflang]
|
|
@@ -301,8 +302,8 @@ function service(args) {
|
|
|
301
302
|
|
|
302
303
|
function seo(args) {
|
|
303
304
|
const command = args[0];
|
|
304
|
-
const scripts = { performance: "maggie_performance.py", images: "maggie_images.py", sitemap: "maggie_sitemap.py" };
|
|
305
|
-
if (!scripts[command]) throw new Error("seo command must be performance, images, or
|
|
305
|
+
const scripts = { performance: "maggie_performance.py", images: "maggie_images.py", sitemap: "maggie_sitemap.py", indexnow: "maggie_indexnow.py" };
|
|
306
|
+
if (!scripts[command]) throw new Error("seo command must be performance, images, sitemap, or indexnow");
|
|
306
307
|
workflowCli(scripts[command], args.slice(1));
|
|
307
308
|
}
|
|
308
309
|
|
|
@@ -27,7 +27,7 @@ locale/revision tasks are skipped; failed tasks retry from checkpoints. Run one
|
|
|
27
27
|
worker per project. Outputs stay draft and non-indexable. Existing host HTTP
|
|
28
28
|
and scheduler integrations still need separate implementation and tests.
|
|
29
29
|
|
|
30
|
-
Use the stable CLI to initialize a local blog contract, ingest versioned local
|
|
30
|
+
Use the stable CLI to initialize a local blog contract, ingest versioned local
|
|
31
31
|
content, validate lifecycle invariants, publish with an actor and reason, and
|
|
32
32
|
generate route/feed artifacts. Read the host framework and database contract
|
|
33
33
|
before adding public routes. The host project owns rendering, persistence
|
|
@@ -35,19 +35,23 @@ credentials, and deployment; this skill owns normalized blog semantics.
|
|
|
35
35
|
|
|
36
36
|
```bash
|
|
37
37
|
maggie blog init --project . --base-path /our-blogs --confirm
|
|
38
|
-
maggie blog ingest --project . --source local --input content/posts.json --confirm
|
|
39
|
-
maggie blog validate --project .
|
|
40
|
-
maggie blog
|
|
38
|
+
maggie blog ingest --project . --source local --input content/posts.json --confirm
|
|
39
|
+
maggie blog validate --project .
|
|
40
|
+
maggie blog check-gate --project .
|
|
41
|
+
maggie blog approve --project . --slug example-post --actor reviewer --reason "reviewed" --confirm
|
|
42
|
+
maggie blog publish --project . --slug example-post --actor owner --reason "approved" --confirm
|
|
41
43
|
maggie blog sitemap --project .
|
|
42
44
|
maggie blog settings --project .
|
|
43
45
|
maggie blog rollback --project . --confirm
|
|
44
46
|
```
|
|
45
47
|
|
|
46
|
-
Posts are draft-first.
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
48
|
+
Posts are draft-first. `maggie blog check-gate` reports the publication policy;
|
|
49
|
+
the default requires `requireReview=true`, disables auto-publish, and forces an
|
|
50
|
+
explicit `approve` transition before `publish`. Stable `contentId` is the
|
|
51
|
+
ingest identity and a published slug must not change during a rewrite.
|
|
52
|
+
Search/sort views are not indexable; drafts never appear in public routes, RSS,
|
|
53
|
+
or sitemap output. Provider keys remain server-side. Public publication and
|
|
54
|
+
migrations always require explicit confirmation.
|
|
51
55
|
|
|
52
56
|
To initialize native front-end pages from the approved local UI guideline,
|
|
53
57
|
run `maggie-design`:
|
|
@@ -200,10 +200,25 @@ maggie dash sections bindings-validate --sections-file .maggie/sections.json \
|
|
|
200
200
|
--references-file .maggie/reference-inventory.json
|
|
201
201
|
maggie dash sections idempotency-validate --contract-file .maggie/reconcile-contract.json
|
|
202
202
|
|
|
203
|
-
#
|
|
204
|
-
maggie dash
|
|
203
|
+
# Validate API body/response shapes before a client writes against an endpoint:
|
|
204
|
+
maggie dash api-contract --spec .maggie/openapi.json
|
|
205
|
+
|
|
206
|
+
# Classify every published page exactly once. `pageKind` describes the public
|
|
207
|
+
# page family; `kind`/`rendererKind` describes how it is rendered. Keep these
|
|
208
|
+
# axes separate and require explicit source coverage when database rows do not
|
|
209
|
+
# contain code-rendered routes:
|
|
210
|
+
maggie dash inventory --pages-file .maggie/published-pages.json \
|
|
211
|
+
--require-page-kinds --require-source-coverage
|
|
205
212
|
```
|
|
206
213
|
|
|
214
|
+
The API contract checker fails closed when a body or 2xx response schema is
|
|
215
|
+
missing. Intentional empty bodies must be explicit (`x-maggie-empty-request-body`
|
|
216
|
+
or `x-maggie-empty-response`) so a consumer does not guess field names.
|
|
217
|
+
Inventory reports `pageKindCounts` for service, variant, category hub,
|
|
218
|
+
ordinary, and blog routes, plus `sourceCoverage`. A host adapter must include
|
|
219
|
+
code-rendered routes in the input or declare coverage incomplete; a zero
|
|
220
|
+
database-row count is not evidence that no public routes exist.
|
|
221
|
+
|
|
207
222
|
The catalogue declares purpose, usage, placement, repeatability and layout
|
|
208
223
|
limits, renderer-owned examples, and the shape of repeated entries. A repeat
|
|
209
224
|
may contain an object (`title`, `body`, `href`, and so on), not just a count;
|
|
@@ -59,6 +59,11 @@ If the change date is unknown, omit `lastmod`. An empty content-type does not
|
|
|
59
59
|
need a sitemap chunk in the sitemap index; serving an empty endpoint and
|
|
60
60
|
advertising it are separate decisions.
|
|
61
61
|
|
|
62
|
+
Validation also reports a freshness-distribution warning when a large sample
|
|
63
|
+
collapses onto one date (especially today). Treat that as a provenance review,
|
|
64
|
+
not as a reason to rewrite dates: verify the content-change event and omit
|
|
65
|
+
unknown dates. Strict semantic validation promotes the warning to a failure.
|
|
66
|
+
|
|
62
67
|
Generated sitemap XML uses the conventional readable shape by default: one
|
|
63
68
|
`<url>`/`<sitemap>` entry per block, UTF-8 XML, and date-only `lastmod` evidence
|
|
64
69
|
rendered as a full UTC W3C datetime. The plan also exposes the response
|
|
@@ -174,6 +179,15 @@ maggie seo sitemap validate --plan docs/sitemap-plan.json
|
|
|
174
179
|
# After an approved host adapter apply, rollback uses its exact backup manifest.
|
|
175
180
|
maggie seo sitemap rollback --backup-manifest .maggie-sitemap-backups/<plan>/backup-manifest.json \
|
|
176
181
|
--public-dir public --confirm
|
|
182
|
+
|
|
183
|
+
# Change-driven IndexNow: only pass URLs whose rendered content changed.
|
|
184
|
+
maggie seo indexnow key-check --public-dir public --key-file <key>.txt --key <key>
|
|
185
|
+
maggie seo indexnow plan --origin https://example.com \
|
|
186
|
+
--changed-urls-file .maggie/changed-urls.json \
|
|
187
|
+
--state-file .maggie/indexnow-state.json --key <key> \
|
|
188
|
+
--output .maggie/indexnow-plan.json
|
|
189
|
+
maggie seo indexnow submit --plan .maggie/indexnow-plan.json \
|
|
190
|
+
--state-file .maggie/indexnow-state.json --confirm
|
|
177
191
|
```
|
|
178
192
|
|
|
179
193
|
Only confirmed image variants may enter `srcset`; `apply` requires an explicit
|
|
@@ -182,6 +196,13 @@ omit empty chunks from the sitemap index, enforce absolute same-origin URLs, and
|
|
|
182
196
|
record redirects for removed sitemap files. Read the [image and sitemap PRD](https://github.com/TOPY-AI-LTD/ai-cmo-skills/blob/main/docs/image-sitemap-structure-prd.md)
|
|
183
197
|
for adapter and rollback rules.
|
|
184
198
|
|
|
199
|
+
IndexNow is opt-in and complements, rather than replaces, the sitemap. The
|
|
200
|
+
host supplies a changed-URL set and a public verification-key file. The plan
|
|
201
|
+
holds back URLs accepted within 24 hours, deduplicates same-origin URLs, and
|
|
202
|
+
records accepted state only after HTTP 200/202. A 403/429/5xx or network error
|
|
203
|
+
is retryable and must never make the editor save fail; do not retry unchanged
|
|
204
|
+
URLs in a loop.
|
|
205
|
+
|
|
185
206
|
For a deterministic technical smoke check, run:
|
|
186
207
|
|
|
187
208
|
```bash
|
|
@@ -27,6 +27,8 @@ def main() -> int:
|
|
|
27
27
|
inspect = sub.add_parser("inspect"); inspect.add_argument("--project", type=Path, default=Path.cwd())
|
|
28
28
|
ingest = sub.add_parser("ingest"); ingest.add_argument("--project", type=Path, default=Path.cwd()); ingest.add_argument("--input", type=Path, required=True); ingest.add_argument("--source", default="local"); ingest.add_argument("--confirm", action="store_true")
|
|
29
29
|
validate = sub.add_parser("validate"); validate.add_argument("--project", type=Path, default=Path.cwd())
|
|
30
|
+
gate = sub.add_parser("check-gate", help="report whether generated content requires review before publication"); gate.add_argument("--project", type=Path, default=Path.cwd())
|
|
31
|
+
approve = sub.add_parser("approve"); approve.add_argument("--project", type=Path, default=Path.cwd()); approve.add_argument("--slug", required=True); approve.add_argument("--actor", required=True); approve.add_argument("--reason", required=True); approve.add_argument("--confirm", action="store_true")
|
|
30
32
|
publish = sub.add_parser("publish"); publish.add_argument("--project", type=Path, default=Path.cwd()); publish.add_argument("--slug", required=True); publish.add_argument("--actor", required=True); publish.add_argument("--reason", required=True); publish.add_argument("--confirm", action="store_true")
|
|
31
33
|
sitemap = sub.add_parser("sitemap"); sitemap.add_argument("--project", type=Path, default=Path.cwd())
|
|
32
34
|
settings = sub.add_parser("settings"); settings.add_argument("--project", type=Path, default=Path.cwd())
|
|
@@ -38,7 +40,7 @@ def main() -> int:
|
|
|
38
40
|
print(json.dumps(result, indent=2, ensure_ascii=False))
|
|
39
41
|
return 0 if result["status"] != "error" else 1
|
|
40
42
|
store = BlogStore(args.project.resolve())
|
|
41
|
-
if args.command in {"init", "ingest", "publish", "rollback", "translate-pending"} and not args.confirm:
|
|
43
|
+
if args.command in {"init", "ingest", "publish", "approve", "rollback", "translate-pending"} and not args.confirm:
|
|
42
44
|
print("CONFIRMATION_REQUIRED: rerun with --confirm", file=sys.stderr); return 2
|
|
43
45
|
try:
|
|
44
46
|
if args.command == "init": result = store.init(args.base_path, args.posts_per_page)
|
|
@@ -51,6 +53,8 @@ def main() -> int:
|
|
|
51
53
|
if not isinstance(payload, list): raise ValueError("local input must be a JSON array")
|
|
52
54
|
result = store.ingest(payload, args.source)
|
|
53
55
|
elif args.command == "validate": result = store.validate()
|
|
56
|
+
elif args.command == "check-gate": result = store.review_gate(store.settings())
|
|
57
|
+
elif args.command == "approve": result = store.approve(args.slug, args.actor, args.reason)
|
|
54
58
|
elif args.command == "publish": result = store.publish(args.slug, args.actor, args.reason)
|
|
55
59
|
elif args.command == "settings": result = store.settings()
|
|
56
60
|
elif args.command == "rollback": result = store.rollback(args.backup)
|
|
@@ -25,6 +25,7 @@ from service_variants import ServiceVariantStore # noqa: E402
|
|
|
25
25
|
from maggie_dash_ui import load_and_validate # noqa: E402
|
|
26
26
|
from maggie_sections import catalogue, remap_translations, section_id_migration, validate_registry, copy_notes, validate_values, validate_fanout, reconcile_fields, validate_locale_coverage # noqa: E402
|
|
27
27
|
from maggie_quality import validate_variant_copy, validate_media_uniqueness, classify_inventory, validate_bindings, validate_reconcile_contract # noqa: E402
|
|
28
|
+
from maggie_api_contract import validate_api_contract # noqa: E402
|
|
28
29
|
from route_imports import classify_bindings # noqa: E402
|
|
29
30
|
|
|
30
31
|
|
|
@@ -422,7 +423,14 @@ def command_components_audit(args: argparse.Namespace) -> int:
|
|
|
422
423
|
def command_inventory(args: argparse.Namespace) -> int:
|
|
423
424
|
"""Classify each published page once, with no broad predicate overlap."""
|
|
424
425
|
value = json.loads(Path(args.pages_file).resolve().read_text(encoding="utf-8"))
|
|
425
|
-
result = classify_inventory(value)
|
|
426
|
+
result = classify_inventory(value, require_page_kinds=args.require_page_kinds, require_source_coverage=args.require_source_coverage)
|
|
427
|
+
emit(result)
|
|
428
|
+
return 0 if result["passed"] else 1
|
|
429
|
+
|
|
430
|
+
|
|
431
|
+
def command_api_contract(args: argparse.Namespace) -> int:
|
|
432
|
+
value = json.loads(Path(args.spec).resolve().read_text(encoding="utf-8"))
|
|
433
|
+
result = validate_api_contract(value)
|
|
426
434
|
emit(result)
|
|
427
435
|
return 0 if result["passed"] else 1
|
|
428
436
|
|
|
@@ -572,7 +580,12 @@ def parser() -> argparse.ArgumentParser:
|
|
|
572
580
|
components.set_defaults(func=command_components_audit)
|
|
573
581
|
inventory = sub.add_parser("inventory", help="classify published pages into disjoint band/code-rendered kinds")
|
|
574
582
|
inventory.add_argument("--pages-file", required=True, help="JSON page inventory")
|
|
583
|
+
inventory.add_argument("--require-page-kinds", action="store_true", help="fail when a published page has no pageKind")
|
|
584
|
+
inventory.add_argument("--require-source-coverage", action="store_true", help="fail unless sourceCoverage.complete is true")
|
|
575
585
|
inventory.set_defaults(func=command_inventory)
|
|
586
|
+
api_contract = sub.add_parser("api-contract", help="validate declared request and success-response schemas")
|
|
587
|
+
api_contract.add_argument("--spec", required=True, help="OpenAPI-like JSON document")
|
|
588
|
+
api_contract.set_defaults(func=command_api_contract)
|
|
576
589
|
variant = sub.add_parser("variant", help="manage service variant lifecycle")
|
|
577
590
|
variant_sub = variant.add_subparsers(dest="variant_command", required=True)
|
|
578
591
|
create = variant_sub.add_parser("create"); create.add_argument("--project", default="."); create.add_argument("--service-id", required=True); create.add_argument("--variant-id", required=True); create.add_argument("--variant-type", required=True); create.add_argument("--locale", required=True); create.add_argument("--market", required=True); create.add_argument("--slug", required=True); create.add_argument("--title", required=True); create.add_argument("--facts", required=True); create.add_argument("--source-revision", required=True); create.add_argument("--canonical-variant-id"); create.add_argument("--cluster-link", action="append", default=[]); create.add_argument("--layout-family", default="service-default"); create.add_argument("--confirm", action="store_true")
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Plan and submit change-driven IndexNow notifications."""
|
|
3
|
+
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
import argparse
|
|
7
|
+
import json
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
import sys
|
|
10
|
+
|
|
11
|
+
sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "runtime"))
|
|
12
|
+
from maggie_indexnow import check_key_file, plan_indexnow, submit_plan # noqa: E402
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def read_json(path: Path, default: object) -> object:
|
|
16
|
+
if not path.exists():
|
|
17
|
+
return default
|
|
18
|
+
return json.loads(path.read_text(encoding="utf-8"))
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def write_json(path: Path, value: object) -> None:
|
|
22
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
23
|
+
path.write_text(json.dumps(value, indent=2, ensure_ascii=False) + "\n", encoding="utf-8")
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def main() -> int:
|
|
27
|
+
parser = argparse.ArgumentParser(prog="maggie seo indexnow")
|
|
28
|
+
sub = parser.add_subparsers(dest="command", required=True)
|
|
29
|
+
key_check = sub.add_parser("key-check", help="verify the public ownership key file")
|
|
30
|
+
key_check.add_argument("--public-dir", required=True); key_check.add_argument("--key-file", required=True); key_check.add_argument("--key", required=True)
|
|
31
|
+
plan = sub.add_parser("plan", help="hold back recently accepted URLs and create a submission plan")
|
|
32
|
+
plan.add_argument("--origin", required=True); plan.add_argument("--changed-urls-file", required=True); plan.add_argument("--state-file", required=True); plan.add_argument("--key", required=True); plan.add_argument("--key-location"); plan.add_argument("--guard-hours", type=int, default=24); plan.add_argument("--now"); plan.add_argument("--output", required=True)
|
|
33
|
+
submit = sub.add_parser("submit", help="submit an approved plan; provider failure never updates accepted state")
|
|
34
|
+
submit.add_argument("--plan", required=True); submit.add_argument("--state-file", required=True); submit.add_argument("--endpoint", default="https://api.indexnow.org/indexnow"); submit.add_argument("--confirm", action="store_true")
|
|
35
|
+
args = parser.parse_args()
|
|
36
|
+
try:
|
|
37
|
+
if args.command == "key-check":
|
|
38
|
+
result = check_key_file(Path(args.public_dir).resolve(), args.key_file, args.key)
|
|
39
|
+
print(json.dumps(result, indent=2)); return 0 if result["status"] == "pass" else 1
|
|
40
|
+
if args.command == "plan":
|
|
41
|
+
changed = read_json(Path(args.changed_urls_file), [])
|
|
42
|
+
urls = changed.get("urls", []) if isinstance(changed, dict) else changed
|
|
43
|
+
if not isinstance(urls, list): raise ValueError("changed-urls-file must be a JSON array or {\"urls\": []}")
|
|
44
|
+
result = plan_indexnow(args.origin, urls, read_json(Path(args.state_file), {}), key=args.key, key_location=args.key_location, guard_hours=args.guard_hours, now=args.now)
|
|
45
|
+
write_json(Path(args.output), result)
|
|
46
|
+
print(json.dumps({"status": "planned", "eligible": len(result["eligible"]), "heldBack": len(result["heldBack"]), "rejected": len(result["rejected"]), "output": str(Path(args.output).resolve())}, indent=2)); return 0
|
|
47
|
+
if not args.confirm: raise ValueError("submit requires --confirm")
|
|
48
|
+
plan_value = read_json(Path(args.plan), {})
|
|
49
|
+
if not isinstance(plan_value, dict) or plan_value.get("schemaVersion") != "maggie-indexnow.v1": raise ValueError("invalid IndexNow plan")
|
|
50
|
+
state_path = Path(args.state_file); state = read_json(state_path, {})
|
|
51
|
+
if not isinstance(state, dict): state = {}
|
|
52
|
+
result = submit_plan(plan_value, state, args.endpoint)
|
|
53
|
+
if result["status"] == "accepted": write_json(state_path, state)
|
|
54
|
+
print(json.dumps(result, indent=2)); return 0 if result["status"] in {"accepted", "no-op"} else 1
|
|
55
|
+
except (OSError, ValueError, TypeError, json.JSONDecodeError) as error:
|
|
56
|
+
print(f"maggie-indexnow: {error}", file=sys.stderr); return 1
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
if __name__ == "__main__":
|
|
60
|
+
raise SystemExit(main())
|
|
@@ -0,0 +1,97 @@
|
|
|
1
|
+
"""Provider-neutral request/response contract checks for HTTP APIs."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections.abc import Mapping
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
SCHEMA = "maggie-api-contract.v1"
|
|
10
|
+
METHODS = {"get", "post", "put", "patch", "delete", "head", "options", "trace"}
|
|
11
|
+
BODY_METHODS = {"post", "put", "patch"}
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def _has_schema(value: object) -> bool:
|
|
15
|
+
if not isinstance(value, Mapping):
|
|
16
|
+
return False
|
|
17
|
+
if "$ref" in value:
|
|
18
|
+
return bool(str(value["$ref"]).strip())
|
|
19
|
+
schema = value.get("schema")
|
|
20
|
+
if isinstance(schema, Mapping):
|
|
21
|
+
return _has_schema(schema) or bool(schema)
|
|
22
|
+
content = value.get("content")
|
|
23
|
+
if isinstance(content, Mapping):
|
|
24
|
+
return any(_has_schema(media) for media in content.values())
|
|
25
|
+
return False
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _success_responses(responses: Mapping[str, Any]) -> list[tuple[str, Mapping[str, Any]]]:
|
|
29
|
+
result = []
|
|
30
|
+
for code, response in responses.items():
|
|
31
|
+
label = str(code).upper()
|
|
32
|
+
if label.startswith("2") or label == "2XX":
|
|
33
|
+
result.append((str(code), response if isinstance(response, Mapping) else {}))
|
|
34
|
+
return result
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def validate_api_contract(spec: object) -> dict[str, Any]:
|
|
38
|
+
"""Validate declared body and success response shapes without guessing fields.
|
|
39
|
+
|
|
40
|
+
Empty request/response bodies are valid only when they are explicit (or a
|
|
41
|
+
204 response). This keeps a consumer from inventing fields for an
|
|
42
|
+
under-specified endpoint while allowing intentional command-style APIs.
|
|
43
|
+
"""
|
|
44
|
+
errors: list[dict[str, str]] = []
|
|
45
|
+
operations: list[dict[str, Any]] = []
|
|
46
|
+
paths = spec.get("paths") if isinstance(spec, Mapping) else None
|
|
47
|
+
if not isinstance(paths, Mapping):
|
|
48
|
+
return {"schemaVersion": SCHEMA, "passed": False, "operations": [],
|
|
49
|
+
"errors": [{"code": "paths-missing", "message": "spec.paths must be an object"}]}
|
|
50
|
+
for path, path_item in sorted(paths.items(), key=lambda item: str(item[0])):
|
|
51
|
+
if not isinstance(path_item, Mapping):
|
|
52
|
+
errors.append({"path": str(path), "code": "path-invalid", "message": "path item must be an object"})
|
|
53
|
+
continue
|
|
54
|
+
for method, operation in sorted(path_item.items(), key=lambda item: str(item[0])):
|
|
55
|
+
verb = str(method).casefold()
|
|
56
|
+
if verb not in METHODS:
|
|
57
|
+
continue
|
|
58
|
+
operation_id = str((operation or {}).get("operationId") or f"{verb} {path}") if isinstance(operation, Mapping) else f"{verb} {path}"
|
|
59
|
+
record: dict[str, Any] = {"path": str(path), "method": verb.upper(), "operationId": operation_id}
|
|
60
|
+
if not isinstance(operation, Mapping):
|
|
61
|
+
errors.append({"path": str(path), "method": verb.upper(), "code": "operation-invalid", "message": "operation must be an object"})
|
|
62
|
+
continue
|
|
63
|
+
if verb in BODY_METHODS:
|
|
64
|
+
body = operation.get("requestBody")
|
|
65
|
+
if body is None and not operation.get("x-maggie-empty-request-body"):
|
|
66
|
+
errors.append({"path": str(path), "method": verb.upper(), "code": "request-body-undeclared", "message": "declare requestBody or x-maggie-empty-request-body"})
|
|
67
|
+
record["requestBody"] = "missing"
|
|
68
|
+
elif operation.get("x-maggie-empty-request-body"):
|
|
69
|
+
record["requestBody"] = "explicit-empty"
|
|
70
|
+
elif not _has_schema(body):
|
|
71
|
+
errors.append({"path": str(path), "method": verb.upper(), "code": "request-body-schema-missing", "message": "requestBody must declare content.schema or $ref"})
|
|
72
|
+
record["requestBody"] = "schema-missing"
|
|
73
|
+
else:
|
|
74
|
+
record["requestBody"] = "declared"
|
|
75
|
+
else:
|
|
76
|
+
record["requestBody"] = "not-applicable"
|
|
77
|
+
responses = operation.get("responses")
|
|
78
|
+
if not isinstance(responses, Mapping):
|
|
79
|
+
errors.append({"path": str(path), "method": verb.upper(), "code": "responses-missing", "message": "declare responses with a 2xx success response"})
|
|
80
|
+
record["successResponse"] = "missing"
|
|
81
|
+
else:
|
|
82
|
+
successes = _success_responses(responses)
|
|
83
|
+
if not successes:
|
|
84
|
+
errors.append({"path": str(path), "method": verb.upper(), "code": "success-response-missing", "message": "declare at least one 2xx response"})
|
|
85
|
+
record["successResponse"] = "missing"
|
|
86
|
+
else:
|
|
87
|
+
code, response = successes[0]
|
|
88
|
+
if code == "204" or response.get("x-maggie-empty-response"):
|
|
89
|
+
record["successResponse"] = f"{code}:explicit-empty"
|
|
90
|
+
elif not _has_schema(response):
|
|
91
|
+
errors.append({"path": str(path), "method": verb.upper(), "code": "success-response-schema-missing", "message": f"response {code} must declare content.schema, $ref, or x-maggie-empty-response"})
|
|
92
|
+
record["successResponse"] = f"{code}:schema-missing"
|
|
93
|
+
else:
|
|
94
|
+
record["successResponse"] = f"{code}:declared"
|
|
95
|
+
operations.append(record)
|
|
96
|
+
return {"schemaVersion": SCHEMA, "passed": not errors, "operations": operations, "errors": errors,
|
|
97
|
+
"operationCount": len(operations), "errorCount": len(errors)}
|
|
@@ -18,6 +18,8 @@ DEFAULTS = {
|
|
|
18
18
|
"postsPerPage": 9,
|
|
19
19
|
"topicLimit": 12,
|
|
20
20
|
"defaultPostStatus": "draft",
|
|
21
|
+
"requireReview": True,
|
|
22
|
+
"autoPublishEnabled": False,
|
|
21
23
|
"autoPullEnabled": False,
|
|
22
24
|
"pullIntervalMinutes": 120,
|
|
23
25
|
"maxPerRun": 1,
|
|
@@ -142,6 +144,12 @@ class BlogStore:
|
|
|
142
144
|
if not content_id or not title: raise ValueError("each post requires contentId/id and title")
|
|
143
145
|
old = by_content.get(content_id)
|
|
144
146
|
raw_slug = str(item.get("slug") or title)
|
|
147
|
+
initial_status = settings.get("defaultPostStatus", "draft")
|
|
148
|
+
# A provider must not turn an import into an implicit publication.
|
|
149
|
+
# Older projects may still carry defaultPostStatus=published; the
|
|
150
|
+
# review gate makes that configuration visible and safe here.
|
|
151
|
+
if settings.get("requireReview", True) and initial_status != "draft":
|
|
152
|
+
initial_status = "draft"
|
|
145
153
|
post = {
|
|
146
154
|
"schemaVersion": "maggie-blog-post.v1",
|
|
147
155
|
"id": old["id"] if old else f"post:{content_id}",
|
|
@@ -156,7 +164,7 @@ class BlogStore:
|
|
|
156
164
|
"topics": [{"slug": slugify(str(topic)), "label": str(topic)} for topic in item.get("topics", item.get("keywords", []))],
|
|
157
165
|
# Provider input can suggest a status, but cannot bypass the
|
|
158
166
|
# local approval gate. Existing published state is preserved.
|
|
159
|
-
"status": old.get("status",
|
|
167
|
+
"status": old.get("status", initial_status) if old else initial_status,
|
|
160
168
|
"canonicalUrl": item.get("canonicalUrl"),
|
|
161
169
|
"source": {"provider": provider, "revision": str(item.get("revision") or checksum(item))},
|
|
162
170
|
"publishedAt": old.get("publishedAt") if old else item.get("publishedAt"),
|
|
@@ -183,6 +191,8 @@ class BlogStore:
|
|
|
183
191
|
|
|
184
192
|
def validate(self) -> dict:
|
|
185
193
|
settings = self.settings(); posts = self.posts(); errors = []
|
|
194
|
+
gate = self.review_gate(settings)
|
|
195
|
+
errors.extend(f"review gate: {error}" for error in gate["errors"])
|
|
186
196
|
ids = [p.get("contentId") for p in posts]; slugs = [p.get("slug") for p in posts]
|
|
187
197
|
if len(ids) != len(set(ids)): errors.append("duplicate contentId")
|
|
188
198
|
if len(slugs) != len(set(slugs)): errors.append("duplicate slug")
|
|
@@ -192,11 +202,42 @@ class BlogStore:
|
|
|
192
202
|
if post.get("status") == "published" and not post.get("publishedAt"): errors.append(f"{post.get('id')}: publishedAt required")
|
|
193
203
|
return {"valid": not errors, "posts": len(posts), "published": sum(p.get("status") == "published" for p in posts), "settings": settings, "errors": errors}
|
|
194
204
|
|
|
205
|
+
@staticmethod
|
|
206
|
+
def review_gate(settings: dict) -> dict:
|
|
207
|
+
"""Report whether generated content is forced through human review."""
|
|
208
|
+
errors: list[str] = []
|
|
209
|
+
require_review = bool(settings.get("requireReview", True))
|
|
210
|
+
auto_publish = bool(settings.get("autoPublishEnabled", False))
|
|
211
|
+
default_status = str(settings.get("defaultPostStatus", "draft"))
|
|
212
|
+
if auto_publish:
|
|
213
|
+
errors.append("autoPublishEnabled must be false for provider-neutral draft-first publishing")
|
|
214
|
+
if not require_review:
|
|
215
|
+
errors.append("requireReview must be true")
|
|
216
|
+
if default_status == "published":
|
|
217
|
+
errors.append("defaultPostStatus cannot be published")
|
|
218
|
+
return {"schemaVersion": "maggie-blog-review-gate.v1", "passed": not errors,
|
|
219
|
+
"requireReview": require_review, "autoPublishEnabled": auto_publish,
|
|
220
|
+
"defaultPostStatus": default_status, "errors": errors,
|
|
221
|
+
"mutation": False}
|
|
222
|
+
|
|
223
|
+
def approve(self, slug: str, actor: str, reason: str) -> dict:
|
|
224
|
+
if not actor or not reason:
|
|
225
|
+
raise ValueError("actor and reason are required")
|
|
226
|
+
posts = self.posts(); found = next((p for p in posts if p.get("slug") == slug), None)
|
|
227
|
+
if not found: raise ValueError(f"post not found: {slug}")
|
|
228
|
+
if found["status"] not in {"draft", "review"}:
|
|
229
|
+
raise ValueError(f"post cannot approve from {found['status']}")
|
|
230
|
+
found["status"] = "approved"; found["updatedAt"] = now()
|
|
231
|
+
found["lastTransition"] = {"actor": actor, "reason": reason, "at": now()}
|
|
232
|
+
self._backup(); self._write(self.posts_path, posts); return found
|
|
233
|
+
|
|
195
234
|
def publish(self, slug: str, actor: str, reason: str) -> dict:
|
|
196
235
|
if not actor or not reason: raise ValueError("actor and reason are required")
|
|
197
236
|
posts = self.posts(); found = next((p for p in posts if p.get("slug") == slug), None)
|
|
198
237
|
if not found: raise ValueError(f"post not found: {slug}")
|
|
199
238
|
if found["status"] not in {"draft", "review", "approved"}: raise ValueError(f"post cannot publish from {found['status']}")
|
|
239
|
+
if self.settings().get("requireReview", True) and found["status"] != "approved":
|
|
240
|
+
raise ValueError("post requires explicit approval before publish")
|
|
200
241
|
found["status"] = "published"; found["publishedAt"] = found.get("publishedAt") or now(); found["updatedAt"] = now(); found["lastTransition"] = {"actor": actor, "reason": reason, "at": now()}
|
|
201
242
|
self._backup(); self._write(self.posts_path, posts); return found
|
|
202
243
|
|
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
"""Provider-neutral, change-driven IndexNow planning and submission state."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import hashlib
|
|
6
|
+
import json
|
|
7
|
+
from datetime import datetime, timedelta, timezone
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from typing import Any
|
|
10
|
+
from urllib.error import HTTPError, URLError
|
|
11
|
+
from urllib.parse import urlparse
|
|
12
|
+
from urllib.request import Request, urlopen
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
SCHEMA = "maggie-indexnow.v1"
|
|
16
|
+
DEFAULT_GUARD_HOURS = 24
|
|
17
|
+
ACCEPTED = {200, 202}
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _now(value: str | None = None) -> datetime:
|
|
21
|
+
if value:
|
|
22
|
+
return datetime.fromisoformat(value.replace("Z", "+00:00")).astimezone(timezone.utc)
|
|
23
|
+
return datetime.now(timezone.utc)
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _iso(value: datetime) -> str:
|
|
27
|
+
return value.astimezone(timezone.utc).isoformat(timespec="seconds").replace("+00:00", "Z")
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def normalise_urls(origin: str, urls: list[object]) -> tuple[list[str], list[str]]:
|
|
31
|
+
"""Return valid same-origin URLs and explicit rejects."""
|
|
32
|
+
base = origin.rstrip("/")
|
|
33
|
+
origin_host = urlparse(base).netloc
|
|
34
|
+
accepted: set[str] = set()
|
|
35
|
+
rejected: list[str] = []
|
|
36
|
+
for raw in urls:
|
|
37
|
+
value = str(raw or "").strip()
|
|
38
|
+
if not value:
|
|
39
|
+
continue
|
|
40
|
+
candidate = value if value.startswith(("http://", "https://")) else f"{base}/{value.lstrip('/')}"
|
|
41
|
+
parsed = urlparse(candidate)
|
|
42
|
+
if parsed.scheme not in {"http", "https"} or parsed.netloc != origin_host or parsed.fragment:
|
|
43
|
+
rejected.append(value)
|
|
44
|
+
continue
|
|
45
|
+
accepted.add(candidate)
|
|
46
|
+
return sorted(accepted), rejected
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _state_entries(state: object) -> list[dict[str, Any]]:
|
|
50
|
+
if not isinstance(state, dict) or not isinstance(state.get("submissions"), list):
|
|
51
|
+
return []
|
|
52
|
+
return [item for item in state["submissions"] if isinstance(item, dict)]
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def plan_indexnow(origin: str, urls: list[object], state: object | None = None,
|
|
56
|
+
*, key: str, key_location: str | None = None,
|
|
57
|
+
guard_hours: int = DEFAULT_GUARD_HOURS, now: str | None = None) -> dict[str, Any]:
|
|
58
|
+
if not key.strip():
|
|
59
|
+
raise ValueError("IndexNow key is required; it is a public ownership key, not a provider secret")
|
|
60
|
+
eligible, rejected = normalise_urls(origin, urls)
|
|
61
|
+
cutoff = _now(now) - timedelta(hours=guard_hours)
|
|
62
|
+
accepted_at: dict[str, datetime] = {}
|
|
63
|
+
for entry in _state_entries(state):
|
|
64
|
+
if entry.get("statusCode") not in ACCEPTED or not entry.get("acceptedAt"):
|
|
65
|
+
continue
|
|
66
|
+
try:
|
|
67
|
+
stamp = _now(str(entry["acceptedAt"]))
|
|
68
|
+
except ValueError:
|
|
69
|
+
continue
|
|
70
|
+
if stamp >= cutoff:
|
|
71
|
+
accepted_at[str(entry.get("url"))] = stamp
|
|
72
|
+
held_back = [{"url": url, "acceptedAt": _iso(accepted_at[url]), "reason": f"accepted within {guard_hours} hours"}
|
|
73
|
+
for url in eligible if url in accepted_at]
|
|
74
|
+
held_set = {item["url"] for item in held_back}
|
|
75
|
+
return {
|
|
76
|
+
"schemaVersion": SCHEMA,
|
|
77
|
+
"origin": origin.rstrip("/"),
|
|
78
|
+
"key": key.strip(),
|
|
79
|
+
"keyLocation": key_location or f"{origin.rstrip('/')}/{key.strip()}.txt",
|
|
80
|
+
"guardHours": guard_hours,
|
|
81
|
+
"eligible": [url for url in eligible if url not in held_set],
|
|
82
|
+
"heldBack": held_back,
|
|
83
|
+
"rejected": rejected,
|
|
84
|
+
"verification": {"required": True, "status": "not-checked"},
|
|
85
|
+
"plannedAt": _iso(_now(now)),
|
|
86
|
+
"mutation": False,
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def check_key_file(public_dir: Path, key_file: str, key: str) -> dict[str, Any]:
|
|
91
|
+
if not key.strip():
|
|
92
|
+
raise ValueError("key is required")
|
|
93
|
+
relative = Path(key_file)
|
|
94
|
+
if relative.is_absolute() or ".." in relative.parts or len(relative.parts) != 1:
|
|
95
|
+
raise ValueError("key file must be a single filename inside public-dir")
|
|
96
|
+
target = public_dir / relative
|
|
97
|
+
served = target.read_text(encoding="utf-8").strip() if target.exists() else None
|
|
98
|
+
return {"schemaVersion": SCHEMA, "path": str(relative), "exists": target.exists(),
|
|
99
|
+
"matches": served == key.strip(), "status": "pass" if served == key.strip() else "fail"}
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def submit_plan(plan: dict[str, Any], state: dict[str, Any], endpoint: str) -> dict[str, Any]:
|
|
103
|
+
urls = plan.get("eligible") if isinstance(plan.get("eligible"), list) else []
|
|
104
|
+
if not urls:
|
|
105
|
+
return {"schemaVersion": SCHEMA, "status": "no-op", "submitted": 0, "accepted": 0, "retryable": False, "mutation": True}
|
|
106
|
+
payload = {"host": urlparse(str(plan["origin"])).netloc, "key": str(plan["key"]),
|
|
107
|
+
"keyLocation": str(plan["keyLocation"]), "urlList": urls}
|
|
108
|
+
status_code: int | None = None
|
|
109
|
+
failure = None
|
|
110
|
+
try:
|
|
111
|
+
request = Request(endpoint, data=json.dumps(payload).encode("utf-8"), headers={"Content-Type": "application/json; charset=utf-8"}, method="POST")
|
|
112
|
+
with urlopen(request, timeout=30) as response:
|
|
113
|
+
status_code = int(response.status)
|
|
114
|
+
except HTTPError as error:
|
|
115
|
+
status_code = int(error.code)
|
|
116
|
+
failure = "provider returned HTTP status"
|
|
117
|
+
except (URLError, TimeoutError, OSError):
|
|
118
|
+
failure = "provider request failed"
|
|
119
|
+
accepted = status_code in ACCEPTED
|
|
120
|
+
retryable = status_code in {403, 429} or (status_code is not None and status_code >= 500) or failure is not None
|
|
121
|
+
entries = _state_entries(state)
|
|
122
|
+
if accepted:
|
|
123
|
+
stamp = _iso(_now())
|
|
124
|
+
entries.extend({"url": url, "statusCode": status_code, "acceptedAt": stamp, "batchId": hashlib.sha256((stamp + url).encode()).hexdigest()[:12]} for url in urls)
|
|
125
|
+
state.update({"schemaVersion": SCHEMA, "submissions": entries})
|
|
126
|
+
return {"schemaVersion": SCHEMA, "status": "accepted" if accepted else "retryable-failure" if retryable else "failed",
|
|
127
|
+
"submitted": len(urls), "accepted": len(urls) if accepted else 0, "statusCode": status_code,
|
|
128
|
+
"retryable": retryable, "failure": failure, "mutation": True}
|
|
@@ -14,6 +14,7 @@ MEDIA_SCHEMA = "maggie-media-uniqueness.v1"
|
|
|
14
14
|
INVENTORY_SCHEMA = "maggie-page-inventory.v1"
|
|
15
15
|
BINDING_SCHEMA = "maggie-section-bindings.v1"
|
|
16
16
|
IDEMPOTENCY_SCHEMA = "maggie-reconcile-contract.v1"
|
|
17
|
+
PAGE_KINDS = {"service", "variant", "category-hub", "ordinary", "blog-topic", "blog-archive", "blog-article", "homepage", "service-index", "service-category", "redirect", "asset"}
|
|
17
18
|
|
|
18
19
|
|
|
19
20
|
def _items(value: object, *keys: str) -> list[dict[str, Any]]:
|
|
@@ -116,8 +117,10 @@ def validate_media_uniqueness(value: object, *, across_siblings: bool = False) -
|
|
|
116
117
|
return {"schemaVersion": MEDIA_SCHEMA, "passed": not errors, "duplicates": errors, "pages": len(pages), "acrossSiblings": across_siblings}
|
|
117
118
|
|
|
118
119
|
|
|
119
|
-
def classify_inventory(value: object) -> dict[str, Any]:
|
|
120
|
+
def classify_inventory(value: object, *, require_page_kinds: bool = False, require_source_coverage: bool = False) -> dict[str, Any]:
|
|
120
121
|
pages = _items(value, "pages", "items", "routes")
|
|
122
|
+
if isinstance(value, dict):
|
|
123
|
+
pages += _items(value, "codeRenderedPages", "codeRenderedRoutes")
|
|
121
124
|
allowed = {"band", "code-rendered", "redirect", "asset", "unknown"}
|
|
122
125
|
records: list[dict[str, Any]] = []
|
|
123
126
|
errors: list[str] = []
|
|
@@ -127,7 +130,7 @@ def classify_inventory(value: object) -> dict[str, Any]:
|
|
|
127
130
|
if path in seen:
|
|
128
131
|
errors.append(f"duplicate inventory path: {path}")
|
|
129
132
|
seen.add(path)
|
|
130
|
-
explicit = str(page.get("kind") or page.get("classification") or "").casefold()
|
|
133
|
+
explicit = str(page.get("rendererKind") or page.get("rendering") or page.get("kind") or page.get("classification") or "").casefold()
|
|
131
134
|
if explicit in {"code", "code-rendered", "component", "source"}:
|
|
132
135
|
kind = "code-rendered"
|
|
133
136
|
elif explicit in {"band", "section", "sections"}:
|
|
@@ -140,11 +143,25 @@ def classify_inventory(value: object) -> dict[str, Any]:
|
|
|
140
143
|
kind = "code-rendered"
|
|
141
144
|
else:
|
|
142
145
|
kind = "unknown"
|
|
143
|
-
|
|
146
|
+
raw_page_kind = page.get("pageKind") or page.get("pageType")
|
|
147
|
+
if not raw_page_kind and str(page.get("kind") or "").casefold() in PAGE_KINDS:
|
|
148
|
+
raw_page_kind = page.get("kind")
|
|
149
|
+
page_kind = str(raw_page_kind or "").casefold().replace("_", "-")
|
|
150
|
+
records.append({"path": path, "kind": kind, "pageKind": page_kind or None})
|
|
144
151
|
if kind not in allowed or kind == "unknown":
|
|
145
152
|
errors.append(f"{path}: cannot classify published page")
|
|
153
|
+
if page_kind and page_kind not in PAGE_KINDS:
|
|
154
|
+
errors.append(f"{path}: unknown pageKind {page_kind}")
|
|
155
|
+
if require_page_kinds and not page_kind:
|
|
156
|
+
errors.append(f"{path}: pageKind is required")
|
|
146
157
|
counts = dict(sorted(Counter(item["kind"] for item in records).items()))
|
|
147
|
-
|
|
158
|
+
page_kind_counts = dict(sorted(Counter(item["pageKind"] for item in records if item["pageKind"]).items()))
|
|
159
|
+
coverage = value.get("sourceCoverage") if isinstance(value, dict) else None
|
|
160
|
+
if not isinstance(coverage, dict):
|
|
161
|
+
coverage = {"status": "unverified", "complete": False, "sources": []}
|
|
162
|
+
if require_source_coverage and coverage.get("complete") is not True:
|
|
163
|
+
errors.append("sourceCoverage.complete must be true when source coverage is required")
|
|
164
|
+
return {"schemaVersion": INVENTORY_SCHEMA, "passed": not errors and len(records) == len(seen), "records": records, "counts": counts, "pageKindCounts": page_kind_counts, "pageKinds": sorted(PAGE_KINDS), "bandCount": counts.get("band", 0), "codeRenderedCount": counts.get("code-rendered", 0), "errors": errors, "disjoint": True, "sourceCoverage": coverage}
|
|
148
165
|
|
|
149
166
|
|
|
150
167
|
def _reference_values(references: object) -> dict[str, set[str]]:
|
|
@@ -15,6 +15,8 @@ HARD_BYTES_LIMIT = 52_428_800
|
|
|
15
15
|
DEFAULT_CHUNK_TARGET = 500
|
|
16
16
|
EXCLUDED_PATHS = re.compile(r"/(search|find|login|draft|preview)(/|$)", re.I)
|
|
17
17
|
W3C_UTC_DATETIME = re.compile(r"^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}Z$")
|
|
18
|
+
FRESHNESS_MIN_SAMPLE = 10
|
|
19
|
+
FRESHNESS_CONCENTRATION_THRESHOLD = 0.75
|
|
18
20
|
|
|
19
21
|
|
|
20
22
|
def absolute_url(url: str, origin: str) -> bool:
|
|
@@ -100,6 +102,51 @@ def xml_file(urls: list[dict[str, str]]) -> str:
|
|
|
100
102
|
return '<?xml version="1.0" encoding="UTF-8"?>\n<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9" xmlns:xhtml="http://www.w3.org/1999/xhtml">\n' + body + ('\n' if body else '') + '</urlset>\n'
|
|
101
103
|
|
|
102
104
|
|
|
105
|
+
def freshness_report(routes: list[dict[str, str]], today: str | None = None) -> dict[str, object]:
|
|
106
|
+
"""Detect a suspiciously uniform lastmod distribution.
|
|
107
|
+
|
|
108
|
+
A structurally valid sitemap can still lose crawler trust when a migration
|
|
109
|
+
stamps nearly every URL with the same operational date. This is a warning
|
|
110
|
+
by default so unknown or intentionally batched content is not silently
|
|
111
|
+
rewritten; strict semantic validation promotes it to an error.
|
|
112
|
+
"""
|
|
113
|
+
dates: list[str] = []
|
|
114
|
+
for route in routes:
|
|
115
|
+
value = route.get("lastmod")
|
|
116
|
+
if not value:
|
|
117
|
+
continue
|
|
118
|
+
normalised = conventional_lastmod(str(value))
|
|
119
|
+
if W3C_UTC_DATETIME.fullmatch(normalised):
|
|
120
|
+
dates.append(normalised[:10])
|
|
121
|
+
counts: dict[str, int] = {}
|
|
122
|
+
for date in dates:
|
|
123
|
+
counts[date] = counts.get(date, 0) + 1
|
|
124
|
+
dominant_date, dominant_count = (max(counts.items(), key=lambda item: (item[1], item[0]))
|
|
125
|
+
if counts else (None, 0))
|
|
126
|
+
sample = len(dates)
|
|
127
|
+
ratio = dominant_count / sample if sample else 0.0
|
|
128
|
+
current = today or datetime.now(timezone.utc).date().isoformat()
|
|
129
|
+
suspicious = bool(sample >= FRESHNESS_MIN_SAMPLE and ratio >= FRESHNESS_CONCENTRATION_THRESHOLD)
|
|
130
|
+
warnings: list[str] = []
|
|
131
|
+
if suspicious:
|
|
132
|
+
date_label = "today's date" if dominant_date == current else str(dominant_date)
|
|
133
|
+
warnings.append(
|
|
134
|
+
f"lastmod distribution is concentrated on {date_label} "
|
|
135
|
+
f"({dominant_count}/{sample}, {ratio:.0%}); verify content-change provenance"
|
|
136
|
+
)
|
|
137
|
+
return {
|
|
138
|
+
"status": "warn" if warnings else "pass",
|
|
139
|
+
"datedRoutes": sample,
|
|
140
|
+
"distinctDates": len(counts),
|
|
141
|
+
"dominantDate": dominant_date,
|
|
142
|
+
"dominantCount": dominant_count,
|
|
143
|
+
"dominantRatio": round(ratio, 4),
|
|
144
|
+
"today": current,
|
|
145
|
+
"todayCount": counts.get(current, 0),
|
|
146
|
+
"warnings": warnings,
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
|
|
103
150
|
def build_plan(routes: list[dict[str, str]], origin: str, content_types: set[str], chunk_target: int = DEFAULT_CHUNK_TARGET, previous: dict | None = None) -> dict:
|
|
104
151
|
if chunk_target < 1 or chunk_target > HARD_URL_LIMIT:
|
|
105
152
|
raise ValueError("chunk target must be between 1 and 50000")
|
|
@@ -134,6 +181,7 @@ def validate_plan_data(plan: dict, strict_semantic: bool = False) -> dict:
|
|
|
134
181
|
if not urlparse(origin).scheme or not urlparse(origin).netloc:
|
|
135
182
|
errors.append("origin must be an absolute HTTP(S) URL")
|
|
136
183
|
expected_urls = []
|
|
184
|
+
all_routes: list[dict[str, str]] = []
|
|
137
185
|
for group in plan.get("groups", []):
|
|
138
186
|
for chunk in group.get("chunks", []):
|
|
139
187
|
expected_urls.append(chunk.get("url"))
|
|
@@ -142,6 +190,7 @@ def validate_plan_data(plan: dict, strict_semantic: bool = False) -> dict:
|
|
|
142
190
|
if chunk.get("bytes", 0) > HARD_BYTES_LIMIT:
|
|
143
191
|
errors.append(f"chunk exceeds byte limit: {chunk.get('filename')}")
|
|
144
192
|
for route in chunk.get("routes", []):
|
|
193
|
+
all_routes.append(route)
|
|
145
194
|
if route.get("indexable") is False or route.get("searchable") is False:
|
|
146
195
|
errors.append(f"non-indexable/searchable route was emitted: {route.get('url')}")
|
|
147
196
|
if route.get("canonicalUrl") and route.get("canonicalUrl") != route.get("url"):
|
|
@@ -187,6 +236,8 @@ def validate_plan_data(plan: dict, strict_semantic: bool = False) -> dict:
|
|
|
187
236
|
errors.append("sitemap index is not valid XML")
|
|
188
237
|
if not expected_urls:
|
|
189
238
|
warnings.append("no sitemap chunks generated; empty content types are not advertised")
|
|
239
|
+
freshness = freshness_report(all_routes)
|
|
240
|
+
warnings.extend(freshness["warnings"])
|
|
190
241
|
if strict_semantic:
|
|
191
242
|
for group in plan.get("groups", []):
|
|
192
243
|
for chunk in group.get("chunks", []):
|
|
@@ -195,7 +246,7 @@ def validate_plan_data(plan: dict, strict_semantic: bool = False) -> dict:
|
|
|
195
246
|
if "contentWords" not in route: warnings.append(f"semantic content evidence missing: {route.get('url')}")
|
|
196
247
|
if warnings:
|
|
197
248
|
errors.extend(warnings)
|
|
198
|
-
return {"status": "pass" if not errors else "fail", "errors": errors, "warnings": warnings}
|
|
249
|
+
return {"status": "pass" if not errors else "fail", "errors": errors, "warnings": warnings, "freshness": freshness}
|
|
199
250
|
|
|
200
251
|
|
|
201
252
|
def agent_files(routes: list[dict[str, str]], origin: str, locale: str = "en") -> dict[str, str]:
|