@topy-ai/maggie 0.7.15 → 0.7.17
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +31 -7
- package/README.zh-TW.md +2 -1
- package/bin/maggie.js +4 -4
- package/bundled-skills/maggie-blog/SKILL.md +11 -1
- package/bundled-skills/maggie-dash/SKILL.md +5 -0
- package/bundled-skills/maggie-ops/SKILL.md +8 -0
- package/bundled-skills/maggie-seo-geo/SKILL.md +29 -1
- package/bundled-tools/clis/maggie.py +1 -1
- package/bundled-tools/clis/maggie_blog.py +14 -2
- package/bundled-tools/clis/maggie_head_tags.py +35 -0
- package/bundled-tools/clis/maggie_indexnow.py +6 -3
- package/bundled-tools/clis/maggie_ops.py +12 -0
- package/bundled-tools/clis/maggie_social_cards.py +45 -0
- package/bundled-tools/clis/site_audit.py +2 -2
- package/bundled-tools/runtime/maggie_blog.py +85 -0
- package/bundled-tools/runtime/maggie_favicon.py +86 -0
- package/bundled-tools/runtime/maggie_head_tags.py +101 -0
- package/bundled-tools/runtime/maggie_indexnow.py +52 -1
- package/bundled-tools/runtime/maggie_quality.py +6 -1
- package/bundled-tools/runtime/maggie_social_cards.py +165 -0
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -80,9 +80,13 @@ The current package also includes reusable safeguards from the latest feedback
|
|
|
80
80
|
review: `maggie dash api-contract` checks declared request and 2xx response
|
|
81
81
|
shapes; `maggie blog check-gate`/`approve` enforces review before publish;
|
|
82
82
|
sitemap validation flags suspiciously uniform `lastmod` dates; `maggie seo
|
|
83
|
-
indexnow` plans only changed same-origin URLs
|
|
84
|
-
|
|
85
|
-
|
|
83
|
+
indexnow` plans only changed same-origin URLs, verifies the deployed key route
|
|
84
|
+
with an optional negative control, and holds back URLs accepted in the last 24
|
|
85
|
+
hours. `maggie seo social-cards` checks per-page OG image dimensions and format;
|
|
86
|
+
`maggie seo head-tags` reports metadata drift across rendered shells;
|
|
87
|
+
`maggie ops favicon-check` verifies the served `/favicon.ico` and declared icon.
|
|
88
|
+
Dash inventory separates renderer kind from public page kind while reporting
|
|
89
|
+
source coverage for code-rendered routes.
|
|
86
90
|
|
|
87
91
|
For Google integrations, validate a redacted provider matrix before reporting
|
|
88
92
|
access. The command fails closed on unknown scopes, missing Ads prerequisites,
|
|
@@ -145,7 +149,10 @@ maggie service ... # import, sync, generate, validate
|
|
|
145
149
|
maggie seo performance ... # sampled PageSpeed/CWV report and baseline
|
|
146
150
|
maggie seo images ... # inventory, variants, confirmation, validate
|
|
147
151
|
maggie seo sitemap ... # typed/semantic plan, agent-files, apply, rollback
|
|
148
|
-
maggie seo indexnow ... # changed URLs, key check,
|
|
152
|
+
maggie seo indexnow ... # changed URLs, deployed key check, 24h guard
|
|
153
|
+
maggie seo social-cards ... # per-page og:image format/dimension audit
|
|
154
|
+
maggie seo head-tags ... # rendered-shell head metadata drift audit
|
|
155
|
+
maggie ops favicon-check ... # served favicon behaviour check
|
|
149
156
|
maggie deployment | migration | release | analytics | schedule
|
|
150
157
|
maggie migration identity --identity-file FILE [--expected-file FILE]
|
|
151
158
|
maggie deployment canary --asset URL=SHA256 --render-report report.json
|
|
@@ -199,6 +206,19 @@ Read the [performance PRD](https://github.com/TOPY-AI-LTD/ai-cmo-skills/blob/mai
|
|
|
199
206
|
and [image/sitemap PRD](https://github.com/TOPY-AI-LTD/ai-cmo-skills/blob/main/docs/image-sitemap-structure-prd.md)
|
|
200
207
|
for adapter, privacy, apply, rollback, and release gates.
|
|
201
208
|
|
|
209
|
+
For a database-backed blog, `maggie blog check-gate` accepts a sanitized host
|
|
210
|
+
settings adapter so review policy can be checked without copying database
|
|
211
|
+
credentials into a local store:
|
|
212
|
+
|
|
213
|
+
```bash
|
|
214
|
+
maggie blog check-gate --project . \
|
|
215
|
+
--adapter-command '["node", "scripts/read-blog-settings.mjs"]'
|
|
216
|
+
```
|
|
217
|
+
|
|
218
|
+
The adapter is an argv array, receives a versioned request on stdin, and must
|
|
219
|
+
print only the three review-policy fields. Provider stderr is suppressed and a
|
|
220
|
+
missing local store without an adapter fails closed.
|
|
221
|
+
|
|
202
222
|
### Design and deployment release gates
|
|
203
223
|
|
|
204
224
|
Check source-to-runtime icon coverage before shipping a design:
|
|
@@ -247,8 +267,8 @@ artifact schemas.
|
|
|
247
267
|
Recommended upgrade sequence for the current release:
|
|
248
268
|
|
|
249
269
|
```bash
|
|
250
|
-
npx @topy-ai/maggie@0.7.
|
|
251
|
-
npx @topy-ai/maggie@0.7.
|
|
270
|
+
npx @topy-ai/maggie@0.7.17 update --project . --force
|
|
271
|
+
npx @topy-ai/maggie@0.7.17 cleanup --project .
|
|
252
272
|
```
|
|
253
273
|
|
|
254
274
|
Maintainers should pass npm credentials through the repository helper, never
|
|
@@ -258,7 +278,11 @@ as a command-line argument:
|
|
|
258
278
|
node scripts/publish-npm.mjs --maggie-env-file ../.env
|
|
259
279
|
```
|
|
260
280
|
|
|
261
|
-
The 0.7.
|
|
281
|
+
The 0.7.17 workflow adds sanitized database-backed blog gate adapters, deployed
|
|
282
|
+
IndexNow key verification, social-card and cross-shell head-tag audits, required
|
|
283
|
+
Open Graph image coverage, and behavioural favicon verification. The 0.7.16
|
|
284
|
+
workflow tightens the review-gate exit code and adds retry-path regression
|
|
285
|
+
coverage. The 0.7.15 workflow added API contracts, blog review gates, sitemap freshness,
|
|
262
286
|
change-driven IndexNow, and page-kind/source-coverage inventory. It also keeps
|
|
263
287
|
the general `maggie-qa-workflow` skill and `maggie qa`
|
|
264
288
|
CLI for scenario manifests, secret-free browser evidence metadata, test/fix/
|
package/README.zh-TW.md
CHANGED
|
@@ -8,7 +8,7 @@ Codex、Claude Code 與相容的 coding agents。
|
|
|
8
8
|
## 安裝
|
|
9
9
|
|
|
10
10
|
```bash
|
|
11
|
-
npx @topy-ai/maggie@0.7.
|
|
11
|
+
npx @topy-ai/maggie@0.7.16 init --agent all
|
|
12
12
|
npx @topy-ai/maggie doctor --project .
|
|
13
13
|
```
|
|
14
14
|
|
|
@@ -25,6 +25,7 @@ maggie doctor --project . --require-bootstrap --strict
|
|
|
25
25
|
deployment、memory、feedback 和 MaggieDash。內容先 draft/review,外部寫入、
|
|
26
26
|
publish 與 production deployment 需要明確確認。
|
|
27
27
|
|
|
28
|
+
0.7.16 補強 review gate 的 CI exit code 與 IndexNow retry regression;
|
|
28
29
|
0.7.15 也加入 API schema contract、blog review gate、sitemap freshness
|
|
29
30
|
warning、change-driven IndexNow 與 page-kind/source-coverage inventory。
|
|
30
31
|
|
package/bin/maggie.js
CHANGED
|
@@ -113,7 +113,7 @@ Usage:
|
|
|
113
113
|
maggie api lifecycle --project PATH [--execute --allow-quota]
|
|
114
114
|
maggie memory <init|list|search|context|add|record-error|transition|export> --project PATH
|
|
115
115
|
maggie localization <extract|plan|generate|preview|validate|review|publish|stale|glossary> [options]
|
|
116
|
-
maggie seo performance|images|sitemap|indexnow [options] (sitemap supports strict validate and agent-files)
|
|
116
|
+
maggie seo performance|images|sitemap|indexnow|social-cards|head-tags [options] (sitemap supports strict validate and agent-files)
|
|
117
117
|
maggie feedback <collect|preview|submit|list> [options]
|
|
118
118
|
maggie qa <start|record|summary|export> [options]
|
|
119
119
|
maggie site-audit URL [--crawl] [--access-log FILE] [--require-sitemap-request] [--languages en-GB,es-MX,ja-JP] [--check-hreflang]
|
|
@@ -121,7 +121,7 @@ Usage:
|
|
|
121
121
|
maggie site-audit URL --crawl --baseline FILE
|
|
122
122
|
maggie browser-audit URL --browse PATH --output DIR --required SELECTOR [--sticky SELECTOR]
|
|
123
123
|
maggie verification coverage --contract FILE --evidence FILE
|
|
124
|
-
maggie ops audit|preflight|verify|lockfiles|seed-manifest|google-capabilities --project PATH
|
|
124
|
+
maggie ops audit|preflight|verify|lockfiles|seed-manifest|google-capabilities|favicon-check --project PATH
|
|
125
125
|
maggie ops google-capabilities --project PATH --report FILE [--output FILE]
|
|
126
126
|
maggie ops preflight --project PATH --write
|
|
127
127
|
|
|
@@ -302,8 +302,8 @@ function service(args) {
|
|
|
302
302
|
|
|
303
303
|
function seo(args) {
|
|
304
304
|
const command = args[0];
|
|
305
|
-
const scripts = { performance: "maggie_performance.py", images: "maggie_images.py", sitemap: "maggie_sitemap.py", indexnow: "maggie_indexnow.py" };
|
|
306
|
-
if (!scripts[command]) throw new Error("seo command must be performance, images, sitemap, or
|
|
305
|
+
const scripts = { performance: "maggie_performance.py", images: "maggie_images.py", sitemap: "maggie_sitemap.py", indexnow: "maggie_indexnow.py", "social-cards": "maggie_social_cards.py", "head-tags": "maggie_head_tags.py" };
|
|
306
|
+
if (!scripts[command]) throw new Error("seo command must be performance, images, sitemap, indexnow, social-cards, or head-tags");
|
|
307
307
|
workflowCli(scripts[command], args.slice(1));
|
|
308
308
|
}
|
|
309
309
|
|
|
@@ -38,6 +38,9 @@ maggie blog init --project . --base-path /our-blogs --confirm
|
|
|
38
38
|
maggie blog ingest --project . --source local --input content/posts.json --confirm
|
|
39
39
|
maggie blog validate --project .
|
|
40
40
|
maggie blog check-gate --project .
|
|
41
|
+
# Database-backed hosts can supply a sanitized policy through an adapter.
|
|
42
|
+
maggie blog check-gate --project . \
|
|
43
|
+
--adapter-command '["node", "scripts/read-blog-settings.mjs"]'
|
|
41
44
|
maggie blog approve --project . --slug example-post --actor reviewer --reason "reviewed" --confirm
|
|
42
45
|
maggie blog publish --project . --slug example-post --actor owner --reason "approved" --confirm
|
|
43
46
|
maggie blog sitemap --project .
|
|
@@ -51,7 +54,14 @@ explicit `approve` transition before `publish`. Stable `contentId` is the
|
|
|
51
54
|
ingest identity and a published slug must not change during a rewrite.
|
|
52
55
|
Search/sort views are not indexable; drafts never appear in public routes, RSS,
|
|
53
56
|
or sitemap output. Provider keys remain server-side. Public publication and
|
|
54
|
-
migrations always require explicit confirmation.
|
|
57
|
+
migrations always require explicit confirmation.
|
|
58
|
+
|
|
59
|
+
The adapter runs as an argv array without a shell, receives a versioned request
|
|
60
|
+
on stdin, and must print only an object containing
|
|
61
|
+
`defaultPostStatus`/`default_post_status`, `requireReview`/`require_review`,
|
|
62
|
+
and `autoPublishEnabled`/`auto_publish`. Provider stderr is suppressed. Use
|
|
63
|
+
`--settings-file` for a reviewed sanitized snapshot. A missing local store with
|
|
64
|
+
neither input fails closed.
|
|
55
65
|
|
|
56
66
|
To initialize native front-end pages from the approved local UI guideline,
|
|
57
67
|
run `maggie-design`:
|
|
@@ -219,6 +219,11 @@ ordinary, and blog routes, plus `sourceCoverage`. A host adapter must include
|
|
|
219
219
|
code-rendered routes in the input or declare coverage incomplete; a zero
|
|
220
220
|
database-row count is not evidence that no public routes exist.
|
|
221
221
|
|
|
222
|
+
When a host renders multiple shells, export a normalized head manifest and run
|
|
223
|
+
`maggie seo head-tags audit`. This is the reusable boundary for detecting icon
|
|
224
|
+
MIME drift, OG-image fallback drift, and verification-tag coverage; MaggieDash
|
|
225
|
+
does not infer these values from a route filename or from one sample page.
|
|
226
|
+
|
|
222
227
|
The catalogue declares purpose, usage, placement, repeatability and layout
|
|
223
228
|
limits, renderer-owned examples, and the shape of repeated entries. A repeat
|
|
224
229
|
may contain an object (`title`, `body`, `href`, and so on), not just a count;
|
|
@@ -42,6 +42,8 @@ python3 tools/clis/maggie_ops.py --project . audit
|
|
|
42
42
|
python3 tools/clis/maggie_ops.py --project . preflight --write
|
|
43
43
|
python3 tools/clis/maggie_ops.py --project . status
|
|
44
44
|
python3 tools/clis/maggie_ops.py --project . verify
|
|
45
|
+
python3 tools/clis/maggie_ops.py --project . favicon-check \
|
|
46
|
+
--origin https://example.com
|
|
45
47
|
python3 tools/clis/maggie_ops.py --project . record sitemap-match \
|
|
46
48
|
--dry-run --quota-impact quota --idempotency-key match-2026-08-29-001
|
|
47
49
|
```
|
|
@@ -52,6 +54,12 @@ python3 tools/clis/maggie_ops.py --project . record sitemap-match \
|
|
|
52
54
|
approved operation but does not call a remote provider; provider mutations
|
|
53
55
|
remain behind the server-side API adapter and its execute gate.
|
|
54
56
|
|
|
57
|
+
`favicon-check` is a deployed behavioural check. It fetches `/favicon.ico`
|
|
58
|
+
and the declared `<link rel="icon">` (or an explicit `--declared-url`), then
|
|
59
|
+
requires HTTP 200, a readable square image, and an engine-supported ICO, PNG,
|
|
60
|
+
or GIF response. A logo-shaped source filename is only an unverified source
|
|
61
|
+
candidate; it is never evidence that the public favicon route works.
|
|
62
|
+
|
|
55
63
|
If the user does not name a mode, inspect the project and propose the smallest
|
|
56
64
|
mode that satisfies the request. Do not rebuild the public blog or change its
|
|
57
65
|
framework just to add Ops.
|
|
@@ -181,13 +181,23 @@ maggie seo sitemap rollback --backup-manifest .maggie-sitemap-backups/<plan>/bac
|
|
|
181
181
|
--public-dir public --confirm
|
|
182
182
|
|
|
183
183
|
# Change-driven IndexNow: only pass URLs whose rendered content changed.
|
|
184
|
-
maggie seo indexnow key-check --public-dir public --key-file <key>.txt --key <key>
|
|
184
|
+
maggie seo indexnow key-check --public-dir public --key-file <key>.txt --key <key> \
|
|
185
|
+
--key-url https://example.com/<key>.txt \
|
|
186
|
+
--wrong-key-url https://example.com/not-a-key.txt
|
|
185
187
|
maggie seo indexnow plan --origin https://example.com \
|
|
186
188
|
--changed-urls-file .maggie/changed-urls.json \
|
|
187
189
|
--state-file .maggie/indexnow-state.json --key <key> \
|
|
188
190
|
--output .maggie/indexnow-plan.json
|
|
189
191
|
maggie seo indexnow submit --plan .maggie/indexnow-plan.json \
|
|
190
192
|
--state-file .maggie/indexnow-state.json --confirm
|
|
193
|
+
|
|
194
|
+
# Resolve each page's og:image and inspect its image contract.
|
|
195
|
+
maggie seo social-cards audit --urls-file .maggie/public-urls.json \
|
|
196
|
+
--output .maggie/social-cards.json
|
|
197
|
+
|
|
198
|
+
# Compare normalized head metadata emitted by every rendered shell.
|
|
199
|
+
maggie seo head-tags audit --pages-file .maggie/rendered-head.json \
|
|
200
|
+
--output .maggie/head-tags.json
|
|
191
201
|
```
|
|
192
202
|
|
|
193
203
|
Only confirmed image variants may enter `srcset`; `apply` requires an explicit
|
|
@@ -203,6 +213,24 @@ records accepted state only after HTTP 200/202. A 403/429/5xx or network error
|
|
|
203
213
|
is retryable and must never make the editor save fail; do not retry unchanged
|
|
204
214
|
URLs in a loop.
|
|
205
215
|
|
|
216
|
+
The complete IndexNow key gate checks both the local public file and the
|
|
217
|
+
deployed URL body. Add a wrong-key URL as a negative control when the host
|
|
218
|
+
route is expected to return 404; response bodies are never printed.
|
|
219
|
+
|
|
220
|
+
Social-card auditing uses a `300x157` minimum for a large summary card and
|
|
221
|
+
accepts JPEG, PNG, GIF, and ICO responses. It reports missing, unreachable,
|
|
222
|
+
undersized, unreadable, or unsupported images and warns when one source is
|
|
223
|
+
used for at least 75% of a sample of three or more pages. Page-specific
|
|
224
|
+
first-band images remain an editorial choice and should be declared by the
|
|
225
|
+
host manifest.
|
|
226
|
+
|
|
227
|
+
Head-tag auditing consumes a host-produced JSON manifest with `url`, `shell`,
|
|
228
|
+
and normalized `tags`. It compares structural declarations for icons,
|
|
229
|
+
`og:image`, image MIME type, verification presence, robots, and Twitter cards;
|
|
230
|
+
route-specific title and canonical values are intentionally excluded from
|
|
231
|
+
shell drift. It reports disagreements instead of guessing which shell is
|
|
232
|
+
correct.
|
|
233
|
+
|
|
206
234
|
For a deterministic technical smoke check, run:
|
|
207
235
|
|
|
208
236
|
```bash
|
|
@@ -206,7 +206,7 @@ def detect(root: Path) -> dict:
|
|
|
206
206
|
asset_contract = {
|
|
207
207
|
"redirects": found("present", "redirects file") if any(name in names for name in {"_redirects", "redirects.json", "redirects.csv"}) else missing("no redirects manifest detected"),
|
|
208
208
|
"security_headers": found("present", name) if (name := next((name for name in {"_headers", "headers.json", "vercel.json", "netlify.toml"} if name in names), None)) else missing("no deployment/header policy detected"),
|
|
209
|
-
"favicon_or_brand":
|
|
209
|
+
"favicon_or_brand": {"status": "Unverified", "value": "source asset candidate found; verify the served icon with `maggie ops favicon-check`", "evidence": [path]} if (path := next((path for path in implementation_paths if any(token in path.lower() for token in {"favicon", "logo", "brand"}) and Path(path).suffix in {".svg", ".png", ".ico", ".webp"}), None)) else missing("no favicon or brand asset detected"),
|
|
210
210
|
"social_image": found("present", path) if (path := next((path for path in implementation_paths if any(token in path.lower() for token in {"og", "social", "twitter"}) and Path(path).suffix in {".jpg", ".jpeg", ".png", ".webp"}), None)) else missing("no social image asset detected"),
|
|
211
211
|
}
|
|
212
212
|
content_model = {
|
|
@@ -9,7 +9,7 @@ import sys
|
|
|
9
9
|
from pathlib import Path
|
|
10
10
|
|
|
11
11
|
sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "runtime"))
|
|
12
|
-
from maggie_blog import BlogStore # noqa: E402
|
|
12
|
+
from maggie_blog import BlogStore, review_settings_from_adapter # noqa: E402
|
|
13
13
|
from localization_runner import process_adapter
|
|
14
14
|
from integration_state import integration_state # noqa: E402
|
|
15
15
|
|
|
@@ -27,7 +27,7 @@ def main() -> int:
|
|
|
27
27
|
inspect = sub.add_parser("inspect"); inspect.add_argument("--project", type=Path, default=Path.cwd())
|
|
28
28
|
ingest = sub.add_parser("ingest"); ingest.add_argument("--project", type=Path, default=Path.cwd()); ingest.add_argument("--input", type=Path, required=True); ingest.add_argument("--source", default="local"); ingest.add_argument("--confirm", action="store_true")
|
|
29
29
|
validate = sub.add_parser("validate"); validate.add_argument("--project", type=Path, default=Path.cwd())
|
|
30
|
-
gate = sub.add_parser("check-gate", help="report whether generated content requires review before publication"); gate.add_argument("--project", type=Path, default=Path.cwd())
|
|
30
|
+
gate = sub.add_parser("check-gate", help="report whether generated content requires review before publication"); gate.add_argument("--project", type=Path, default=Path.cwd()); gate.add_argument("--settings-file", type=Path, help="sanitized host settings JSON"); gate.add_argument("--adapter-command", help="trusted provider JSON argv array that prints sanitized settings JSON"); gate.add_argument("--timeout", type=int, default=120)
|
|
31
31
|
approve = sub.add_parser("approve"); approve.add_argument("--project", type=Path, default=Path.cwd()); approve.add_argument("--slug", required=True); approve.add_argument("--actor", required=True); approve.add_argument("--reason", required=True); approve.add_argument("--confirm", action="store_true")
|
|
32
32
|
publish = sub.add_parser("publish"); publish.add_argument("--project", type=Path, default=Path.cwd()); publish.add_argument("--slug", required=True); publish.add_argument("--actor", required=True); publish.add_argument("--reason", required=True); publish.add_argument("--confirm", action="store_true")
|
|
33
33
|
sitemap = sub.add_parser("sitemap"); sitemap.add_argument("--project", type=Path, default=Path.cwd())
|
|
@@ -39,6 +39,16 @@ def main() -> int:
|
|
|
39
39
|
result = integration_state(configured=args.configured, consent_required=args.consent_required, consent=args.consent, authorized=args.authorized, error=args.error)
|
|
40
40
|
print(json.dumps(result, indent=2, ensure_ascii=False))
|
|
41
41
|
return 0 if result["status"] != "error" else 1
|
|
42
|
+
if args.command == "check-gate" and (args.settings_file or args.adapter_command):
|
|
43
|
+
try:
|
|
44
|
+
adapter_command = json.loads(args.adapter_command) if args.adapter_command else None
|
|
45
|
+
settings, source = review_settings_from_adapter(args.project.resolve(), args.settings_file.resolve() if args.settings_file else None, adapter_command, args.timeout)
|
|
46
|
+
result = BlogStore.review_gate(settings)
|
|
47
|
+
result["source"] = source
|
|
48
|
+
except (OSError, ValueError, json.JSONDecodeError) as error:
|
|
49
|
+
print(f"maggie-blog: {error}", file=sys.stderr); return 1
|
|
50
|
+
print(json.dumps(result, indent=2, ensure_ascii=False))
|
|
51
|
+
return 0 if result.get("passed") else 1
|
|
42
52
|
store = BlogStore(args.project.resolve())
|
|
43
53
|
if args.command in {"init", "ingest", "publish", "approve", "rollback", "translate-pending"} and not args.confirm:
|
|
44
54
|
print("CONFIRMATION_REQUIRED: rerun with --confirm", file=sys.stderr); return 2
|
|
@@ -64,6 +74,8 @@ def main() -> int:
|
|
|
64
74
|
print(json.dumps(result, indent=2, ensure_ascii=False))
|
|
65
75
|
if args.command in {"ingest", "translate-pending"} and result.get("status") == "partial":
|
|
66
76
|
return 1
|
|
77
|
+
if args.command == "check-gate":
|
|
78
|
+
return 0 if result.get("passed") else 1
|
|
67
79
|
return 0 if args.command != "validate" or result.get("valid") else 1
|
|
68
80
|
|
|
69
81
|
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Compare normalized rendered head tags across site shells."""
|
|
3
|
+
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
import argparse
|
|
7
|
+
import json
|
|
8
|
+
import sys
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
|
|
11
|
+
sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "runtime"))
|
|
12
|
+
from maggie_head_tags import audit_head_tags, read_pages # noqa: E402
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def main() -> int:
|
|
16
|
+
parser = argparse.ArgumentParser(prog="maggie seo head-tags")
|
|
17
|
+
sub = parser.add_subparsers(dest="command", required=True)
|
|
18
|
+
audit = sub.add_parser("audit", help="report head-tag disagreements across rendered shells")
|
|
19
|
+
audit.add_argument("--pages-file", type=Path, required=True, help="JSON page manifest with url, shell, and tags")
|
|
20
|
+
audit.add_argument("--output")
|
|
21
|
+
args = parser.parse_args()
|
|
22
|
+
try:
|
|
23
|
+
result = audit_head_tags(read_pages(args.pages_file))
|
|
24
|
+
if args.output:
|
|
25
|
+
output = Path(args.output).resolve(); output.parent.mkdir(parents=True, exist_ok=True)
|
|
26
|
+
output.write_text(json.dumps(result, indent=2, ensure_ascii=False) + "\n", encoding="utf-8")
|
|
27
|
+
print(json.dumps(result, indent=2, ensure_ascii=False))
|
|
28
|
+
return 0 if result["passed"] else 1
|
|
29
|
+
except (OSError, ValueError, json.JSONDecodeError) as error:
|
|
30
|
+
print(f"maggie-head-tags: {error}", file=sys.stderr)
|
|
31
|
+
return 1
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
if __name__ == "__main__":
|
|
35
|
+
raise SystemExit(main())
|
|
@@ -9,7 +9,7 @@ from pathlib import Path
|
|
|
9
9
|
import sys
|
|
10
10
|
|
|
11
11
|
sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "runtime"))
|
|
12
|
-
from maggie_indexnow import check_key_file, plan_indexnow, submit_plan # noqa: E402
|
|
12
|
+
from maggie_indexnow import check_key_file, check_key_http, plan_indexnow, submit_plan # noqa: E402
|
|
13
13
|
|
|
14
14
|
|
|
15
15
|
def read_json(path: Path, default: object) -> object:
|
|
@@ -26,8 +26,8 @@ def write_json(path: Path, value: object) -> None:
|
|
|
26
26
|
def main() -> int:
|
|
27
27
|
parser = argparse.ArgumentParser(prog="maggie seo indexnow")
|
|
28
28
|
sub = parser.add_subparsers(dest="command", required=True)
|
|
29
|
-
key_check = sub.add_parser("key-check", help="verify
|
|
30
|
-
key_check.add_argument("--public-dir", required=True); key_check.add_argument("--key-file", required=True); key_check.add_argument("--key", required=True)
|
|
29
|
+
key_check = sub.add_parser("key-check", help="verify local and deployed public ownership key evidence")
|
|
30
|
+
key_check.add_argument("--public-dir", required=True); key_check.add_argument("--key-file", required=True); key_check.add_argument("--key", required=True); key_check.add_argument("--key-url", help="deployed key URL to fetch"); key_check.add_argument("--wrong-key-url", help="negative-control URL expected to return HTTP 404"); key_check.add_argument("--timeout", type=int, default=15)
|
|
31
31
|
plan = sub.add_parser("plan", help="hold back recently accepted URLs and create a submission plan")
|
|
32
32
|
plan.add_argument("--origin", required=True); plan.add_argument("--changed-urls-file", required=True); plan.add_argument("--state-file", required=True); plan.add_argument("--key", required=True); plan.add_argument("--key-location"); plan.add_argument("--guard-hours", type=int, default=24); plan.add_argument("--now"); plan.add_argument("--output", required=True)
|
|
33
33
|
submit = sub.add_parser("submit", help="submit an approved plan; provider failure never updates accepted state")
|
|
@@ -36,6 +36,9 @@ def main() -> int:
|
|
|
36
36
|
try:
|
|
37
37
|
if args.command == "key-check":
|
|
38
38
|
result = check_key_file(Path(args.public_dir).resolve(), args.key_file, args.key)
|
|
39
|
+
if args.key_url:
|
|
40
|
+
result["http"] = check_key_http(args.key_url, args.key, args.wrong_key_url, args.timeout)
|
|
41
|
+
if result["http"]["status"] != "pass": result["status"] = "fail"
|
|
39
42
|
print(json.dumps(result, indent=2)); return 0 if result["status"] == "pass" else 1
|
|
40
43
|
if args.command == "plan":
|
|
41
44
|
changed = read_json(Path(args.changed_urls_file), [])
|
|
@@ -15,6 +15,7 @@ from seed_evidence import validate_manifest
|
|
|
15
15
|
from agent_runtime import preflight as agent_preflight
|
|
16
16
|
from restart_gate import validate as validate_restart_gate
|
|
17
17
|
from google_capabilities import validate_report as validate_google_capability_report
|
|
18
|
+
from maggie_favicon import check_favicon
|
|
18
19
|
|
|
19
20
|
|
|
20
21
|
REQUIRED_ROUTES = (
|
|
@@ -181,6 +182,12 @@ def command_google_capabilities(args: argparse.Namespace) -> int:
|
|
|
181
182
|
return 0 if result["passed"] else 1
|
|
182
183
|
|
|
183
184
|
|
|
185
|
+
def command_favicon_check(args: argparse.Namespace) -> int:
|
|
186
|
+
result = check_favicon(args.origin, args.declared_url, args.timeout)
|
|
187
|
+
print(json.dumps(result, indent=2, ensure_ascii=False))
|
|
188
|
+
return 0 if result["passed"] else 1
|
|
189
|
+
|
|
190
|
+
|
|
184
191
|
def main() -> int:
|
|
185
192
|
parser = argparse.ArgumentParser(description=__doc__)
|
|
186
193
|
parser.add_argument("--project", default=".")
|
|
@@ -203,6 +210,11 @@ def main() -> int:
|
|
|
203
210
|
google.add_argument("--report", required=True)
|
|
204
211
|
google.add_argument("--output")
|
|
205
212
|
google.set_defaults(func=command_google_capabilities)
|
|
213
|
+
favicon = sub.add_parser("favicon-check", help="fetch /favicon.ico and the declared icon to verify behaviour")
|
|
214
|
+
favicon.add_argument("--origin", required=True)
|
|
215
|
+
favicon.add_argument("--declared-url")
|
|
216
|
+
favicon.add_argument("--timeout", type=int, default=15)
|
|
217
|
+
favicon.set_defaults(func=command_favicon_check)
|
|
206
218
|
record = sub.add_parser("record", help="record an explicitly authorised operation")
|
|
207
219
|
record.add_argument("operation", choices=("sync", "sitemap-match", "rewrite-queue", "rewrite-approve", "publish", "migration"))
|
|
208
220
|
record.add_argument("--dry-run", action="store_true")
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Audit per-page Open Graph social-card assets."""
|
|
3
|
+
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
import argparse
|
|
7
|
+
import json
|
|
8
|
+
import sys
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
|
|
11
|
+
sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "runtime"))
|
|
12
|
+
from maggie_social_cards import audit_cards, pages_from_urls, read_pages # noqa: E402
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def main() -> int:
|
|
16
|
+
parser = argparse.ArgumentParser(prog="maggie seo social-cards")
|
|
17
|
+
sub = parser.add_subparsers(dest="command", required=True)
|
|
18
|
+
audit = sub.add_parser("audit", help="resolve each page og:image and inspect its image contract")
|
|
19
|
+
sources = audit.add_mutually_exclusive_group(required=True)
|
|
20
|
+
sources.add_argument("--urls-file", type=Path, help="JSON array or {\"urls\": []} of HTML URLs")
|
|
21
|
+
sources.add_argument("--pages-file", type=Path, help="JSON page manifest with url and ogImage fields")
|
|
22
|
+
audit.add_argument("--timeout", type=int, default=15)
|
|
23
|
+
audit.add_argument("--output")
|
|
24
|
+
args = parser.parse_args()
|
|
25
|
+
try:
|
|
26
|
+
pages = read_pages(args.pages_file) if args.pages_file else None
|
|
27
|
+
if args.urls_file:
|
|
28
|
+
value = json.loads(args.urls_file.read_text(encoding="utf-8"))
|
|
29
|
+
urls = value.get("urls") if isinstance(value, dict) else value
|
|
30
|
+
if not isinstance(urls, list) or any(not isinstance(url, str) or not url.strip() for url in urls):
|
|
31
|
+
raise ValueError("urls file must be a JSON array or {\"urls\": []} of nonempty strings")
|
|
32
|
+
pages = pages_from_urls(urls, args.timeout)
|
|
33
|
+
result = audit_cards(pages or [], timeout=args.timeout)
|
|
34
|
+
if args.output:
|
|
35
|
+
output = Path(args.output).resolve(); output.parent.mkdir(parents=True, exist_ok=True)
|
|
36
|
+
output.write_text(json.dumps(result, indent=2, ensure_ascii=False) + "\n", encoding="utf-8")
|
|
37
|
+
print(json.dumps(result, indent=2, ensure_ascii=False))
|
|
38
|
+
return 0 if result["passed"] else 1
|
|
39
|
+
except (OSError, ValueError, json.JSONDecodeError) as error:
|
|
40
|
+
print(f"maggie-social-cards: {error}", file=sys.stderr)
|
|
41
|
+
return 1
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
if __name__ == "__main__":
|
|
45
|
+
raise SystemExit(main())
|
|
@@ -144,7 +144,7 @@ def audit_page(url: str, html: str, status: int, content_type: str, expected_lan
|
|
|
144
144
|
"canonical": bool(canonical) and urlparse(canonical).fragment == "",
|
|
145
145
|
"description": bool(page.meta.get("description")),
|
|
146
146
|
"locale": valid_locale(page.lang) and (not expected_languages or parse_locale(page.lang)[0] in expected_languages or page.lang in expected_languages),
|
|
147
|
-
"open_graph": all(page.meta.get(key) for key in ("og:title", "og:description", "og:url")),
|
|
147
|
+
"open_graph": all(page.meta.get(key) for key in ("og:title", "og:description", "og:url", "og:image")),
|
|
148
148
|
"twitter_card": page.meta.get("twitter:card") in {"summary", "summary_large_image"},
|
|
149
149
|
"jsonld": page.jsonld > 0 and all(item is not None for item in page.jsonld_values),
|
|
150
150
|
"entity_jsonld": any(isinstance(item, dict) and item.get("@type") and (item.get("url") or item.get("@id")) for item in page.jsonld_values),
|
|
@@ -312,7 +312,7 @@ def main() -> int:
|
|
|
312
312
|
checks["canonical"] = {"ok": bool(canonical) and urlparse(canonical).fragment == "", "value": canonical}
|
|
313
313
|
checks["description"] = {"ok": bool(page.meta.get("description")), "value": page.meta.get("description", "")}
|
|
314
314
|
checks["locale"] = {"ok": valid_locale(page.lang) and (not expected_languages or parse_locale(page.lang)[0] in expected_languages or page.lang in expected_languages), "value": page.lang, "language": parse_locale(page.lang)[0], "markets": sorted(markets)}
|
|
315
|
-
checks["open_graph"] = {"ok": all(page.meta.get(key) for key in ("og:title", "og:description", "og:url")), "fields": {key: bool(page.meta.get(key)) for key in ("og:title", "og:description", "og:url")}}
|
|
315
|
+
checks["open_graph"] = {"ok": all(page.meta.get(key) for key in ("og:title", "og:description", "og:url", "og:image")), "fields": {key: bool(page.meta.get(key)) for key in ("og:title", "og:description", "og:url", "og:image")}}
|
|
316
316
|
checks["twitter_card"] = {"ok": page.meta.get("twitter:card") in {"summary", "summary_large_image"}, "value": page.meta.get("twitter:card", "")}
|
|
317
317
|
checks["jsonld"] = {"ok": page.jsonld > 0 and all(item is not None for item in page.jsonld_values), "count": page.jsonld}
|
|
318
318
|
checks["entity_jsonld"] = {"ok": any(isinstance(item, dict) and item.get("@type") and (item.get("url") or item.get("@id")) for item in page.jsonld_values), "count": page.jsonld}
|
|
@@ -6,6 +6,7 @@ import hashlib
|
|
|
6
6
|
import json
|
|
7
7
|
import re
|
|
8
8
|
import shutil
|
|
9
|
+
import subprocess
|
|
9
10
|
from datetime import datetime, timezone
|
|
10
11
|
from pathlib import Path
|
|
11
12
|
from localization_runner import checkpoint, generate
|
|
@@ -44,6 +45,90 @@ def checksum(value: object) -> str:
|
|
|
44
45
|
return "sha256:" + hashlib.sha256(json.dumps(value, sort_keys=True, ensure_ascii=False).encode()).hexdigest()
|
|
45
46
|
|
|
46
47
|
|
|
48
|
+
def normalise_review_settings(value: object) -> dict:
|
|
49
|
+
"""Map a host settings record to the small review-gate contract.
|
|
50
|
+
|
|
51
|
+
Hosts may store settings in PostgreSQL or another provider. The adapter
|
|
52
|
+
boundary deliberately accepts only these policy fields; credentials and
|
|
53
|
+
unrelated provider data never enter the report.
|
|
54
|
+
"""
|
|
55
|
+
if not isinstance(value, dict):
|
|
56
|
+
raise ValueError("blog settings adapter must return a JSON object")
|
|
57
|
+
source = value.get("settings") if isinstance(value.get("settings"), dict) else value
|
|
58
|
+
|
|
59
|
+
def first(*keys: str, default: object = None) -> object:
|
|
60
|
+
for key in keys:
|
|
61
|
+
if key in source:
|
|
62
|
+
return source[key]
|
|
63
|
+
return default
|
|
64
|
+
|
|
65
|
+
def as_bool(value: object, default: bool) -> bool:
|
|
66
|
+
if value is None:
|
|
67
|
+
return default
|
|
68
|
+
if isinstance(value, bool):
|
|
69
|
+
return value
|
|
70
|
+
if isinstance(value, str):
|
|
71
|
+
lowered = value.strip().lower()
|
|
72
|
+
if lowered in {"true", "1", "yes", "on"}:
|
|
73
|
+
return True
|
|
74
|
+
if lowered in {"false", "0", "no", "off", ""}:
|
|
75
|
+
return False
|
|
76
|
+
return bool(value)
|
|
77
|
+
|
|
78
|
+
default_status = str(first("defaultPostStatus", "default_post_status", default="draft") or "draft")
|
|
79
|
+
return {
|
|
80
|
+
"schemaVersion": "maggie-blog-settings.v1",
|
|
81
|
+
"defaultPostStatus": default_status,
|
|
82
|
+
"requireReview": as_bool(first("requireReview", "require_review", default=True), True),
|
|
83
|
+
"autoPublishEnabled": as_bool(first("autoPublishEnabled", "auto_publish", "autoPublish", default=False), False),
|
|
84
|
+
}
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def review_settings_from_adapter(project: Path, settings_file: Path | None = None,
|
|
88
|
+
adapter_command: list[str] | None = None,
|
|
89
|
+
timeout: int = 120) -> tuple[dict, str]:
|
|
90
|
+
"""Read sanitized settings from a file or trusted argv adapter.
|
|
91
|
+
|
|
92
|
+
``adapter_command`` is executed without a shell and receives a small
|
|
93
|
+
request on stdin. Its stdout must be the normalized settings object. Error
|
|
94
|
+
output is intentionally suppressed so provider details cannot leak into a
|
|
95
|
+
Maggie report.
|
|
96
|
+
"""
|
|
97
|
+
if settings_file and adapter_command:
|
|
98
|
+
raise ValueError("use either --settings-file or --adapter-command")
|
|
99
|
+
if timeout < 1:
|
|
100
|
+
raise ValueError("adapter timeout must be positive")
|
|
101
|
+
if settings_file:
|
|
102
|
+
try:
|
|
103
|
+
value = json.loads(settings_file.read_text(encoding="utf-8"))
|
|
104
|
+
except json.JSONDecodeError:
|
|
105
|
+
raise ValueError("settings file must contain a JSON object") from None
|
|
106
|
+
return normalise_review_settings(value), "settings-file"
|
|
107
|
+
if adapter_command is not None:
|
|
108
|
+
if not adapter_command or any(not isinstance(part, str) or not part for part in adapter_command):
|
|
109
|
+
raise ValueError("adapter-command must be a nonempty JSON argv array")
|
|
110
|
+
try:
|
|
111
|
+
result = subprocess.run(
|
|
112
|
+
adapter_command,
|
|
113
|
+
input=json.dumps({"schemaVersion": "maggie-blog-settings-request.v1"}),
|
|
114
|
+
text=True,
|
|
115
|
+
capture_output=True,
|
|
116
|
+
timeout=timeout,
|
|
117
|
+
check=False,
|
|
118
|
+
cwd=project.resolve(),
|
|
119
|
+
)
|
|
120
|
+
except (OSError, subprocess.TimeoutExpired):
|
|
121
|
+
raise OSError("blog settings adapter could not execute or timed out") from None
|
|
122
|
+
if result.returncode:
|
|
123
|
+
raise OSError("blog settings adapter failed; provider output suppressed")
|
|
124
|
+
try:
|
|
125
|
+
value = json.loads(result.stdout)
|
|
126
|
+
except json.JSONDecodeError:
|
|
127
|
+
raise ValueError("blog settings adapter must return a JSON object") from None
|
|
128
|
+
return normalise_review_settings(value), "adapter-command"
|
|
129
|
+
raise ValueError("blog is not initialized; use --settings-file or --adapter-command for a provider-backed blog")
|
|
130
|
+
|
|
131
|
+
|
|
47
132
|
class BlogStore:
|
|
48
133
|
def __init__(self, project: Path) -> None:
|
|
49
134
|
self.root = project / ".maggie" / "blog"
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
"""Behavioural favicon verification for a deployed origin."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from html.parser import HTMLParser
|
|
6
|
+
from typing import Any
|
|
7
|
+
from urllib.error import HTTPError, URLError
|
|
8
|
+
from urllib.parse import urljoin, urlparse
|
|
9
|
+
from urllib.request import Request, urlopen
|
|
10
|
+
|
|
11
|
+
from maggie_social_cards import SUPPORTED_FORMATS, _dimensions
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
SCHEMA = "maggie-favicon.v1"
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class IconLinkParser(HTMLParser):
|
|
18
|
+
def __init__(self) -> None:
|
|
19
|
+
super().__init__()
|
|
20
|
+
self.href = ""
|
|
21
|
+
self.type = ""
|
|
22
|
+
|
|
23
|
+
def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
|
|
24
|
+
if tag != "link":
|
|
25
|
+
return
|
|
26
|
+
data = dict(attrs)
|
|
27
|
+
rel = set(str(data.get("rel") or "").lower().split())
|
|
28
|
+
if rel & {"icon", "shortcut"} and data.get("href") and not self.href:
|
|
29
|
+
self.href = str(data["href"])
|
|
30
|
+
self.type = str(data.get("type") or "")
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _fetch(url: str, timeout: int) -> tuple[int | None, str, bytes, str | None]:
|
|
34
|
+
try:
|
|
35
|
+
request = Request(url, headers={"User-Agent": "Maggie-Favicon-Check/1"})
|
|
36
|
+
with urlopen(request, timeout=timeout) as response:
|
|
37
|
+
raw = response.read(8_000_001)
|
|
38
|
+
if len(raw) > 8_000_000:
|
|
39
|
+
return int(response.status), response.headers.get_content_type(), b"", "response exceeds size limit"
|
|
40
|
+
return int(response.status), response.headers.get_content_type(), raw, None
|
|
41
|
+
except HTTPError as error:
|
|
42
|
+
return int(error.code), "", b"", "HTTP error"
|
|
43
|
+
except (URLError, TimeoutError, OSError):
|
|
44
|
+
return None, "", b"", "request failed"
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def _resource(url: str, timeout: int) -> dict[str, Any]:
|
|
48
|
+
status, header_type, raw, fetch_error = _fetch(url, timeout)
|
|
49
|
+
image_type, width, height = _dimensions(raw, header_type)
|
|
50
|
+
errors: list[str] = []
|
|
51
|
+
if fetch_error or status != 200:
|
|
52
|
+
errors.append(fetch_error or "icon request did not return HTTP 200")
|
|
53
|
+
if image_type not in SUPPORTED_FORMATS:
|
|
54
|
+
errors.append("unsupported favicon format")
|
|
55
|
+
if width is None or height is None:
|
|
56
|
+
errors.append("favicon dimensions could not be read")
|
|
57
|
+
elif width != height:
|
|
58
|
+
errors.append("favicon must be square")
|
|
59
|
+
return {"url": url, "statusCode": status, "format": image_type, "width": width, "height": height, "errors": errors, "passed": not errors}
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def check_favicon(origin: str, declared_url: str | None = None, timeout: int = 15) -> dict[str, Any]:
|
|
63
|
+
parsed = urlparse(origin.rstrip("/"))
|
|
64
|
+
if parsed.scheme not in {"http", "https"} or not parsed.netloc or parsed.fragment:
|
|
65
|
+
raise ValueError("origin must be an absolute HTTP(S) URL without a fragment")
|
|
66
|
+
if timeout < 1:
|
|
67
|
+
raise ValueError("timeout must be positive")
|
|
68
|
+
base = origin.rstrip("/") + "/"
|
|
69
|
+
favicon_url = urljoin(base, "favicon.ico")
|
|
70
|
+
detected: str | None = None
|
|
71
|
+
if not declared_url:
|
|
72
|
+
status, content_type, raw, _ = _fetch(base, timeout)
|
|
73
|
+
if status == 200 and content_type == "text/html":
|
|
74
|
+
parser = IconLinkParser(); parser.feed(raw.decode("utf-8", "replace"))
|
|
75
|
+
if parser.href:
|
|
76
|
+
detected = urljoin(base, parser.href)
|
|
77
|
+
declared = declared_url or detected
|
|
78
|
+
if declared:
|
|
79
|
+
declared = urljoin(base, declared)
|
|
80
|
+
declared_parsed = urlparse(declared)
|
|
81
|
+
if declared_parsed.scheme not in {"http", "https"} or declared_parsed.fragment:
|
|
82
|
+
raise ValueError("declared-url must resolve to an absolute HTTP(S) URL without a fragment")
|
|
83
|
+
ico = _resource(favicon_url, timeout)
|
|
84
|
+
declared_check = _resource(declared, timeout) if declared and declared != favicon_url else None
|
|
85
|
+
passed = ico["passed"] and (declared_check is None or declared_check["passed"])
|
|
86
|
+
return {"schemaVersion": SCHEMA, "origin": origin.rstrip("/"), "favicon": ico, "declaredUrl": declared, "declared": declared_check, "passed": passed, "mutation": False}
|
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
"""Audit normalized head metadata across rendered shells."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
from collections import defaultdict
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from urllib.parse import urlparse
|
|
9
|
+
from typing import Any
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
SCHEMA = "maggie-head-tags.v1"
|
|
13
|
+
STABLE_TAGS = ("icon", "og:image", "og:image:type", "verification", "robots", "twitter:card")
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def _extension(value: object) -> str | None:
|
|
17
|
+
path = urlparse(str(value or "")).path.lower()
|
|
18
|
+
if "." not in path.rsplit("/", 1)[-1]:
|
|
19
|
+
return None
|
|
20
|
+
return "." + path.rsplit("/", 1)[-1].rsplit(".", 1)[-1]
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def _signature(name: str, value: object, tags: dict[str, Any]) -> dict[str, Any]:
|
|
24
|
+
"""Return only safe, structural metadata; never echo verification values."""
|
|
25
|
+
if name == "verification":
|
|
26
|
+
present = bool(value) if not isinstance(value, dict) else bool(value.get("present", value.get("content")))
|
|
27
|
+
return {"present": present}
|
|
28
|
+
if name == "icon":
|
|
29
|
+
data = value if isinstance(value, dict) else {"href": value}
|
|
30
|
+
return {"present": bool(value), "rel": str(data.get("rel") or "icon"), "type": str(data.get("type") or ""), "extension": _extension(data.get("href"))}
|
|
31
|
+
if name == "og:image":
|
|
32
|
+
data = value if isinstance(value, dict) else {"href": value}
|
|
33
|
+
image_type = data.get("type") or tags.get("og:image:type") or ""
|
|
34
|
+
return {"present": bool(value), "type": str(image_type), "extension": _extension(data.get("href") or data.get("content") or value), "fallback": bool(data.get("fallback", tags.get("og:image:fallback", False)))}
|
|
35
|
+
if name == "og:image:type":
|
|
36
|
+
return {"present": bool(value), "type": str(value or "")}
|
|
37
|
+
if isinstance(value, dict):
|
|
38
|
+
return {"present": bool(value), "type": str(value.get("type") or "")}
|
|
39
|
+
return {"present": bool(value), "value": str(value or "")}
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def audit_head_tags(pages: list[dict[str, Any]]) -> dict[str, Any]:
|
|
43
|
+
errors: list[dict[str, str]] = []
|
|
44
|
+
observations: list[dict[str, Any]] = []
|
|
45
|
+
for page in pages:
|
|
46
|
+
if not isinstance(page, dict) or not str(page.get("url") or "").strip():
|
|
47
|
+
errors.append({"reason": "each page requires a URL"})
|
|
48
|
+
continue
|
|
49
|
+
shell = str(page.get("shell") or "").strip()
|
|
50
|
+
if not shell:
|
|
51
|
+
errors.append({"url": str(page["url"]), "reason": "each page requires a shell"})
|
|
52
|
+
continue
|
|
53
|
+
tags = page.get("tags", {})
|
|
54
|
+
if not isinstance(tags, dict):
|
|
55
|
+
errors.append({"url": str(page["url"]), "reason": "tags must be an object"})
|
|
56
|
+
continue
|
|
57
|
+
observations.append({"url": str(page["url"]), "shell": shell, "tags": {name: _signature(name, tags.get(name), tags) for name in STABLE_TAGS}})
|
|
58
|
+
|
|
59
|
+
by_shell: dict[str, list[dict[str, Any]]] = defaultdict(list)
|
|
60
|
+
for observation in observations:
|
|
61
|
+
by_shell[observation["shell"]].append(observation)
|
|
62
|
+
disagreements: list[dict[str, Any]] = []
|
|
63
|
+
|
|
64
|
+
def compare(scope: str, label: str, entries: list[tuple[str, dict[str, Any]]]) -> None:
|
|
65
|
+
for tag in STABLE_TAGS:
|
|
66
|
+
variants: dict[str, dict[str, Any]] = {}
|
|
67
|
+
urls: dict[str, list[str]] = defaultdict(list)
|
|
68
|
+
for identifier, observation in entries:
|
|
69
|
+
signature = observation["tags"][tag]
|
|
70
|
+
key = json.dumps(signature, sort_keys=True, separators=(",", ":"))
|
|
71
|
+
variants[key] = signature
|
|
72
|
+
urls[key].append(observation["url"])
|
|
73
|
+
if len(variants) > 1:
|
|
74
|
+
disagreements.append({"scope": scope, "shell": label if scope == "shell" else None, "tag": tag, "variants": [{"signature": signature, "urls": sorted(urls[key])} for key, signature in sorted(variants.items())]})
|
|
75
|
+
|
|
76
|
+
for shell, shell_pages in sorted(by_shell.items()):
|
|
77
|
+
compare("shell", shell, [(shell, page) for page in shell_pages])
|
|
78
|
+
for tag in STABLE_TAGS:
|
|
79
|
+
entries = [(shell, page) for shell, shell_pages in sorted(by_shell.items()) for page in shell_pages]
|
|
80
|
+
shell_signatures: dict[str, dict[str, Any]] = {}
|
|
81
|
+
shell_urls: dict[str, list[str]] = defaultdict(list)
|
|
82
|
+
for shell, page in entries:
|
|
83
|
+
key = json.dumps(page["tags"][tag], sort_keys=True, separators=(",", ":"))
|
|
84
|
+
shell_signatures[shell] = page["tags"][tag]
|
|
85
|
+
shell_urls[key].append(page["url"])
|
|
86
|
+
unique = {json.dumps(value, sort_keys=True, separators=(",", ":")) for value in shell_signatures.values()}
|
|
87
|
+
if len(unique) > 1:
|
|
88
|
+
disagreements.append({"scope": "across-shells", "shell": None, "tag": tag, "variants": [{"signature": shell_signatures[shell], "shell": shell, "urls": sorted(shell_urls[json.dumps(shell_signatures[shell], sort_keys=True, separators=(",", ":"))])} for shell in sorted(shell_signatures)]})
|
|
89
|
+
|
|
90
|
+
coverage = []
|
|
91
|
+
for shell, shell_pages in sorted(by_shell.items()):
|
|
92
|
+
coverage.append({"shell": shell, "pageCount": len(shell_pages), "tags": {tag: sum(1 for page in shell_pages if page["tags"][tag]["present"]) for tag in STABLE_TAGS}})
|
|
93
|
+
return {"schemaVersion": SCHEMA, "shells": coverage, "observations": observations, "disagreements": disagreements, "errors": errors, "passed": not errors and not disagreements, "mutation": False}
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def read_pages(path: Path) -> list[dict[str, Any]]:
|
|
97
|
+
value = json.loads(path.read_text(encoding="utf-8"))
|
|
98
|
+
pages = value.get("pages") if isinstance(value, dict) else value
|
|
99
|
+
if not isinstance(pages, list):
|
|
100
|
+
raise ValueError("pages file must be a JSON array or {\"pages\": []}")
|
|
101
|
+
return pages
|
|
@@ -99,6 +99,56 @@ def check_key_file(public_dir: Path, key_file: str, key: str) -> dict[str, Any]:
|
|
|
99
99
|
"matches": served == key.strip(), "status": "pass" if served == key.strip() else "fail"}
|
|
100
100
|
|
|
101
101
|
|
|
102
|
+
def _public_url(value: str, label: str) -> str:
|
|
103
|
+
parsed = urlparse(value)
|
|
104
|
+
if parsed.scheme not in {"http", "https"} or not parsed.netloc or parsed.fragment:
|
|
105
|
+
raise ValueError(f"{label} must be an absolute HTTP(S) URL without a fragment")
|
|
106
|
+
return value
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def _fetch_key(url: str, expected: str, timeout: int) -> dict[str, Any]:
|
|
110
|
+
status_code: int | None = None
|
|
111
|
+
matches = False
|
|
112
|
+
try:
|
|
113
|
+
request = Request(url, headers={"User-Agent": "Maggie-IndexNow-Key-Check/1"})
|
|
114
|
+
with urlopen(request, timeout=timeout) as response:
|
|
115
|
+
status_code = int(response.status)
|
|
116
|
+
body = response.read(4097)
|
|
117
|
+
matches = len(body) <= 4096 and body.decode("utf-8", "replace").strip() == expected
|
|
118
|
+
except HTTPError as error:
|
|
119
|
+
status_code = int(error.code)
|
|
120
|
+
except (URLError, TimeoutError, OSError):
|
|
121
|
+
pass
|
|
122
|
+
return {"statusCode": status_code, "matches": matches}
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def check_key_http(key_url: str, key: str, wrong_key_url: str | None = None,
|
|
126
|
+
timeout: int = 15) -> dict[str, Any]:
|
|
127
|
+
"""Verify the deployed key route without exposing its response body."""
|
|
128
|
+
expected = key.strip()
|
|
129
|
+
if not expected:
|
|
130
|
+
raise ValueError("key is required")
|
|
131
|
+
if timeout < 1:
|
|
132
|
+
raise ValueError("timeout must be positive")
|
|
133
|
+
correct_url = _public_url(key_url, "key-url")
|
|
134
|
+
correct = _fetch_key(correct_url, expected, timeout)
|
|
135
|
+
result: dict[str, Any] = {
|
|
136
|
+
"keyUrl": correct_url,
|
|
137
|
+
"correct": correct,
|
|
138
|
+
"wrong": None,
|
|
139
|
+
"status": "pass" if correct["statusCode"] == 200 and correct["matches"] else "fail",
|
|
140
|
+
}
|
|
141
|
+
if wrong_key_url:
|
|
142
|
+
wrong_url = _public_url(wrong_key_url, "wrong-key-url")
|
|
143
|
+
if wrong_url == correct_url:
|
|
144
|
+
raise ValueError("wrong-key-url must differ from key-url")
|
|
145
|
+
wrong = _fetch_key(wrong_url, expected, timeout)
|
|
146
|
+
result["wrong"] = {"url": wrong_url, **wrong, "is404": wrong["statusCode"] == 404}
|
|
147
|
+
if not result["wrong"]["is404"]:
|
|
148
|
+
result["status"] = "fail"
|
|
149
|
+
return result
|
|
150
|
+
|
|
151
|
+
|
|
102
152
|
def submit_plan(plan: dict[str, Any], state: dict[str, Any], endpoint: str) -> dict[str, Any]:
|
|
103
153
|
urls = plan.get("eligible") if isinstance(plan.get("eligible"), list) else []
|
|
104
154
|
if not urls:
|
|
@@ -122,7 +172,8 @@ def submit_plan(plan: dict[str, Any], state: dict[str, Any], endpoint: str) -> d
|
|
|
122
172
|
if accepted:
|
|
123
173
|
stamp = _iso(_now())
|
|
124
174
|
entries.extend({"url": url, "statusCode": status_code, "acceptedAt": stamp, "batchId": hashlib.sha256((stamp + url).encode()).hexdigest()[:12]} for url in urls)
|
|
125
|
-
|
|
175
|
+
if accepted:
|
|
176
|
+
state.update({"schemaVersion": SCHEMA, "submissions": entries})
|
|
126
177
|
return {"schemaVersion": SCHEMA, "status": "accepted" if accepted else "retryable-failure" if retryable else "failed",
|
|
127
178
|
"submitted": len(urls), "accepted": len(urls) if accepted else 0, "statusCode": status_code,
|
|
128
179
|
"retryable": retryable, "failure": failure, "mutation": True}
|
|
@@ -130,7 +130,12 @@ def classify_inventory(value: object, *, require_page_kinds: bool = False, requi
|
|
|
130
130
|
if path in seen:
|
|
131
131
|
errors.append(f"duplicate inventory path: {path}")
|
|
132
132
|
seen.add(path)
|
|
133
|
-
|
|
133
|
+
explicit_kind = str(page.get("kind") or "").casefold()
|
|
134
|
+
# `kind` was historically the renderer field, but the page-kind
|
|
135
|
+
# vocabulary is also accepted for compact host inventories. In that
|
|
136
|
+
# case infer the renderer from sections/source and keep the axes apart.
|
|
137
|
+
renderer_value = page.get("rendererKind") or page.get("rendering") or (None if explicit_kind in PAGE_KINDS else page.get("kind"))
|
|
138
|
+
explicit = str(renderer_value or page.get("classification") or "").casefold()
|
|
134
139
|
if explicit in {"code", "code-rendered", "component", "source"}:
|
|
135
140
|
kind = "code-rendered"
|
|
136
141
|
elif explicit in {"band", "section", "sections"}:
|
|
@@ -0,0 +1,165 @@
|
|
|
1
|
+
"""Provider-neutral Open Graph social-card inspection."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
from collections import Counter
|
|
7
|
+
from html.parser import HTMLParser
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from typing import Any
|
|
10
|
+
from urllib.error import HTTPError, URLError
|
|
11
|
+
from urllib.parse import urljoin, urlparse
|
|
12
|
+
from urllib.request import Request, urlopen
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
SCHEMA = "maggie-social-cards.v1"
|
|
16
|
+
MIN_WIDTH = 300
|
|
17
|
+
MIN_HEIGHT = 157
|
|
18
|
+
SUPPORTED_FORMATS = {"image/jpeg", "image/png", "image/gif", "image/x-icon"}
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class OpenGraphParser(HTMLParser):
|
|
22
|
+
def __init__(self) -> None:
|
|
23
|
+
super().__init__()
|
|
24
|
+
self.values: dict[str, str] = {}
|
|
25
|
+
|
|
26
|
+
def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
|
|
27
|
+
if tag != "meta":
|
|
28
|
+
return
|
|
29
|
+
data = dict(attrs)
|
|
30
|
+
key = (data.get("property") or data.get("name") or "").lower()
|
|
31
|
+
if key in {"og:image", "og:image:type"} and key not in self.values:
|
|
32
|
+
self.values[key] = str(data.get("content") or "").strip()
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _dimensions(raw: bytes, content_type: str) -> tuple[str | None, int | None, int | None]:
|
|
36
|
+
"""Read dimensions from common headers without an image dependency."""
|
|
37
|
+
if raw.startswith(b"\x89PNG\r\n\x1a\n") and len(raw) >= 24:
|
|
38
|
+
return "image/png", int.from_bytes(raw[16:20], "big"), int.from_bytes(raw[20:24], "big")
|
|
39
|
+
if raw[:3] == b"GIF" and len(raw) >= 10:
|
|
40
|
+
return "image/gif", int.from_bytes(raw[6:8], "little"), int.from_bytes(raw[8:10], "little")
|
|
41
|
+
if raw[:4] == b"\x00\x00\x01\x00" and len(raw) >= 22:
|
|
42
|
+
width = raw[6] or 256
|
|
43
|
+
height = raw[7] or 256
|
|
44
|
+
return "image/x-icon", width, height
|
|
45
|
+
if raw[:2] == b"\xff\xd8":
|
|
46
|
+
index = 2
|
|
47
|
+
while index + 9 < len(raw):
|
|
48
|
+
if raw[index] != 0xFF:
|
|
49
|
+
index += 1
|
|
50
|
+
continue
|
|
51
|
+
marker = raw[index + 1]
|
|
52
|
+
index += 2
|
|
53
|
+
if marker in {0xD8, 0xD9}:
|
|
54
|
+
continue
|
|
55
|
+
if index + 2 > len(raw):
|
|
56
|
+
break
|
|
57
|
+
segment_length = int.from_bytes(raw[index:index + 2], "big")
|
|
58
|
+
if segment_length < 2 or index + segment_length > len(raw):
|
|
59
|
+
break
|
|
60
|
+
if marker in set(range(0xC0, 0xC4)) | set(range(0xC5, 0xC8)) | set(range(0xC9, 0xCC)) | set(range(0xCD, 0xD0)):
|
|
61
|
+
if segment_length >= 7:
|
|
62
|
+
return "image/jpeg", int.from_bytes(raw[index + 5:index + 7], "big"), int.from_bytes(raw[index + 3:index + 5], "big")
|
|
63
|
+
index += segment_length
|
|
64
|
+
return "image/jpeg", None, None
|
|
65
|
+
if raw[:4] == b"RIFF" and raw[8:12] == b"WEBP" and len(raw) >= 30:
|
|
66
|
+
if raw[12:16] == b"VP8X":
|
|
67
|
+
width = 1 + int.from_bytes(raw[24:27], "little")
|
|
68
|
+
height = 1 + int.from_bytes(raw[27:30], "little")
|
|
69
|
+
return "image/webp", width, height
|
|
70
|
+
return "image/webp", None, None
|
|
71
|
+
return (content_type.split(";", 1)[0].lower() or None), None, None
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _fetch(url: str, timeout: int, limit: int = 8_000_000) -> tuple[int | None, str, bytes, str | None]:
|
|
75
|
+
try:
|
|
76
|
+
request = Request(url, headers={"User-Agent": "Maggie-Social-Card-Audit/1"})
|
|
77
|
+
with urlopen(request, timeout=timeout) as response:
|
|
78
|
+
raw = response.read(limit + 1)
|
|
79
|
+
return int(response.status), response.headers.get_content_type(), raw[:limit], None if len(raw) <= limit else "response exceeds size limit"
|
|
80
|
+
except HTTPError as error:
|
|
81
|
+
return int(error.code), "", b"", "HTTP error"
|
|
82
|
+
except (URLError, TimeoutError, OSError):
|
|
83
|
+
return None, "", b"", "request failed"
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def _absolute(value: str, base: str) -> str:
|
|
87
|
+
candidate = urljoin(base.rstrip("/") + "/", value)
|
|
88
|
+
parsed = urlparse(candidate)
|
|
89
|
+
if parsed.scheme not in {"http", "https"} or not parsed.netloc or parsed.fragment:
|
|
90
|
+
raise ValueError("og:image must resolve to an absolute HTTP(S) URL without a fragment")
|
|
91
|
+
return candidate
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def audit_cards(pages: list[dict[str, Any]], *, timeout: int = 15,
|
|
95
|
+
min_width: int = MIN_WIDTH, min_height: int = MIN_HEIGHT) -> dict[str, Any]:
|
|
96
|
+
if timeout < 1 or min_width < 1 or min_height < 1:
|
|
97
|
+
raise ValueError("timeout and image dimensions must be positive")
|
|
98
|
+
reports: list[dict[str, Any]] = []
|
|
99
|
+
errors: list[dict[str, str]] = []
|
|
100
|
+
sources: list[str] = []
|
|
101
|
+
for page in pages:
|
|
102
|
+
if not isinstance(page, dict) or not str(page.get("url") or "").strip():
|
|
103
|
+
errors.append({"reason": "page must contain a URL"})
|
|
104
|
+
continue
|
|
105
|
+
page_url = str(page["url"]).strip()
|
|
106
|
+
image = str(page.get("ogImage") or page.get("og:image") or "").strip()
|
|
107
|
+
record: dict[str, Any] = {"url": page_url, "ogImage": image or None, "fallback": bool(page.get("fallback", False)), "errors": []}
|
|
108
|
+
if not image:
|
|
109
|
+
record["errors"].append("missing og:image")
|
|
110
|
+
reports.append(record)
|
|
111
|
+
errors.append({"url": page_url, "reason": "missing og:image"})
|
|
112
|
+
continue
|
|
113
|
+
try:
|
|
114
|
+
image_url = _absolute(image, page_url)
|
|
115
|
+
except ValueError as exc:
|
|
116
|
+
record["errors"].append(str(exc))
|
|
117
|
+
reports.append(record)
|
|
118
|
+
errors.append({"url": page_url, "reason": str(exc)})
|
|
119
|
+
continue
|
|
120
|
+
status, header_type, raw, fetch_error = _fetch(image_url, timeout)
|
|
121
|
+
image_type, width, height = _dimensions(raw, header_type)
|
|
122
|
+
record.update({"imageUrl": image_url, "statusCode": status, "format": image_type, "width": width, "height": height})
|
|
123
|
+
if fetch_error or status != 200:
|
|
124
|
+
record["errors"].append(fetch_error or "image request did not return HTTP 200")
|
|
125
|
+
elif image_type not in SUPPORTED_FORMATS:
|
|
126
|
+
record["errors"].append("unsupported format for large social card")
|
|
127
|
+
elif width is None or height is None:
|
|
128
|
+
record["errors"].append("image dimensions could not be read")
|
|
129
|
+
else:
|
|
130
|
+
if width < min_width or height < min_height:
|
|
131
|
+
record["errors"].append(f"image is smaller than {min_width}x{min_height}")
|
|
132
|
+
reports.append(record)
|
|
133
|
+
sources.append(image_url)
|
|
134
|
+
for reason in record["errors"]:
|
|
135
|
+
errors.append({"url": page_url, "reason": reason})
|
|
136
|
+
|
|
137
|
+
warnings: list[dict[str, Any]] = []
|
|
138
|
+
if len(sources) >= 3:
|
|
139
|
+
source, count = Counter(sources).most_common(1)[0]
|
|
140
|
+
if count / len(sources) >= 0.75:
|
|
141
|
+
warnings.append({"reason": "generic fallback concentration", "imageUrl": source, "count": count, "sampleSize": len(sources), "ratio": round(count / len(sources), 3)})
|
|
142
|
+
return {"schemaVersion": SCHEMA, "threshold": {"minWidth": min_width, "minHeight": min_height, "supportedFormats": sorted(SUPPORTED_FORMATS)}, "pages": reports, "warnings": warnings, "errors": errors, "passed": not errors, "mutation": False}
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def pages_from_urls(urls: list[str], timeout: int) -> list[dict[str, Any]]:
|
|
146
|
+
pages: list[dict[str, Any]] = []
|
|
147
|
+
for page_url in urls:
|
|
148
|
+
status, content_type, raw, fetch_error = _fetch(page_url, timeout, 2_000_000)
|
|
149
|
+
if fetch_error or status != 200 or content_type != "text/html":
|
|
150
|
+
pages.append({"url": page_url, "ogImage": "", "fetchError": fetch_error or "page request did not return HTML"})
|
|
151
|
+
continue
|
|
152
|
+
parser = OpenGraphParser()
|
|
153
|
+
parser.feed(raw.decode("utf-8", "replace"))
|
|
154
|
+
pages.append({"url": page_url, "ogImage": parser.values.get("og:image", "")})
|
|
155
|
+
return pages
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def read_pages(path: Path) -> list[dict[str, Any]]:
|
|
159
|
+
value = json.loads(path.read_text(encoding="utf-8"))
|
|
160
|
+
pages = value.get("pages") if isinstance(value, dict) else value
|
|
161
|
+
if not isinstance(pages, list):
|
|
162
|
+
raise ValueError("pages file must be a JSON array or {\"pages\": []}")
|
|
163
|
+
if any(not isinstance(page, dict) for page in pages):
|
|
164
|
+
raise ValueError("pages file entries must be objects")
|
|
165
|
+
return pages
|