@topy-ai/maggie 0.7.13 → 0.7.15
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +43 -7
- package/README.zh-TW.md +11 -3
- package/bin/maggie.js +8 -5
- package/bundled-skills/README.md +1 -0
- package/bundled-skills/catalog.json +4 -0
- package/bundled-skills/maggie-blog/SKILL.md +13 -9
- package/bundled-skills/maggie-dash/SKILL.md +17 -2
- package/bundled-skills/maggie-qa-workflow/SKILL.md +102 -0
- package/bundled-skills/maggie-seo-geo/SKILL.md +21 -0
- package/bundled-tools/clis/maggie_blog.py +5 -1
- package/bundled-tools/clis/maggie_dash.py +14 -1
- package/bundled-tools/clis/maggie_indexnow.py +60 -0
- package/bundled-tools/clis/maggie_qa_workflow.py +367 -0
- package/bundled-tools/runtime/maggie_api_contract.py +97 -0
- package/bundled-tools/runtime/maggie_blog.py +42 -1
- package/bundled-tools/runtime/maggie_indexnow.py +128 -0
- package/bundled-tools/runtime/maggie_quality.py +21 -4
- package/bundled-tools/runtime/maggie_sitemap.py +52 -1
- package/package.json +1 -1
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
"""Provider-neutral, change-driven IndexNow planning and submission state."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import hashlib
|
|
6
|
+
import json
|
|
7
|
+
from datetime import datetime, timedelta, timezone
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from typing import Any
|
|
10
|
+
from urllib.error import HTTPError, URLError
|
|
11
|
+
from urllib.parse import urlparse
|
|
12
|
+
from urllib.request import Request, urlopen
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
SCHEMA = "maggie-indexnow.v1"
|
|
16
|
+
DEFAULT_GUARD_HOURS = 24
|
|
17
|
+
ACCEPTED = {200, 202}
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _now(value: str | None = None) -> datetime:
|
|
21
|
+
if value:
|
|
22
|
+
return datetime.fromisoformat(value.replace("Z", "+00:00")).astimezone(timezone.utc)
|
|
23
|
+
return datetime.now(timezone.utc)
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _iso(value: datetime) -> str:
|
|
27
|
+
return value.astimezone(timezone.utc).isoformat(timespec="seconds").replace("+00:00", "Z")
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def normalise_urls(origin: str, urls: list[object]) -> tuple[list[str], list[str]]:
|
|
31
|
+
"""Return valid same-origin URLs and explicit rejects."""
|
|
32
|
+
base = origin.rstrip("/")
|
|
33
|
+
origin_host = urlparse(base).netloc
|
|
34
|
+
accepted: set[str] = set()
|
|
35
|
+
rejected: list[str] = []
|
|
36
|
+
for raw in urls:
|
|
37
|
+
value = str(raw or "").strip()
|
|
38
|
+
if not value:
|
|
39
|
+
continue
|
|
40
|
+
candidate = value if value.startswith(("http://", "https://")) else f"{base}/{value.lstrip('/')}"
|
|
41
|
+
parsed = urlparse(candidate)
|
|
42
|
+
if parsed.scheme not in {"http", "https"} or parsed.netloc != origin_host or parsed.fragment:
|
|
43
|
+
rejected.append(value)
|
|
44
|
+
continue
|
|
45
|
+
accepted.add(candidate)
|
|
46
|
+
return sorted(accepted), rejected
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _state_entries(state: object) -> list[dict[str, Any]]:
|
|
50
|
+
if not isinstance(state, dict) or not isinstance(state.get("submissions"), list):
|
|
51
|
+
return []
|
|
52
|
+
return [item for item in state["submissions"] if isinstance(item, dict)]
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def plan_indexnow(origin: str, urls: list[object], state: object | None = None,
|
|
56
|
+
*, key: str, key_location: str | None = None,
|
|
57
|
+
guard_hours: int = DEFAULT_GUARD_HOURS, now: str | None = None) -> dict[str, Any]:
|
|
58
|
+
if not key.strip():
|
|
59
|
+
raise ValueError("IndexNow key is required; it is a public ownership key, not a provider secret")
|
|
60
|
+
eligible, rejected = normalise_urls(origin, urls)
|
|
61
|
+
cutoff = _now(now) - timedelta(hours=guard_hours)
|
|
62
|
+
accepted_at: dict[str, datetime] = {}
|
|
63
|
+
for entry in _state_entries(state):
|
|
64
|
+
if entry.get("statusCode") not in ACCEPTED or not entry.get("acceptedAt"):
|
|
65
|
+
continue
|
|
66
|
+
try:
|
|
67
|
+
stamp = _now(str(entry["acceptedAt"]))
|
|
68
|
+
except ValueError:
|
|
69
|
+
continue
|
|
70
|
+
if stamp >= cutoff:
|
|
71
|
+
accepted_at[str(entry.get("url"))] = stamp
|
|
72
|
+
held_back = [{"url": url, "acceptedAt": _iso(accepted_at[url]), "reason": f"accepted within {guard_hours} hours"}
|
|
73
|
+
for url in eligible if url in accepted_at]
|
|
74
|
+
held_set = {item["url"] for item in held_back}
|
|
75
|
+
return {
|
|
76
|
+
"schemaVersion": SCHEMA,
|
|
77
|
+
"origin": origin.rstrip("/"),
|
|
78
|
+
"key": key.strip(),
|
|
79
|
+
"keyLocation": key_location or f"{origin.rstrip('/')}/{key.strip()}.txt",
|
|
80
|
+
"guardHours": guard_hours,
|
|
81
|
+
"eligible": [url for url in eligible if url not in held_set],
|
|
82
|
+
"heldBack": held_back,
|
|
83
|
+
"rejected": rejected,
|
|
84
|
+
"verification": {"required": True, "status": "not-checked"},
|
|
85
|
+
"plannedAt": _iso(_now(now)),
|
|
86
|
+
"mutation": False,
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def check_key_file(public_dir: Path, key_file: str, key: str) -> dict[str, Any]:
|
|
91
|
+
if not key.strip():
|
|
92
|
+
raise ValueError("key is required")
|
|
93
|
+
relative = Path(key_file)
|
|
94
|
+
if relative.is_absolute() or ".." in relative.parts or len(relative.parts) != 1:
|
|
95
|
+
raise ValueError("key file must be a single filename inside public-dir")
|
|
96
|
+
target = public_dir / relative
|
|
97
|
+
served = target.read_text(encoding="utf-8").strip() if target.exists() else None
|
|
98
|
+
return {"schemaVersion": SCHEMA, "path": str(relative), "exists": target.exists(),
|
|
99
|
+
"matches": served == key.strip(), "status": "pass" if served == key.strip() else "fail"}
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def submit_plan(plan: dict[str, Any], state: dict[str, Any], endpoint: str) -> dict[str, Any]:
|
|
103
|
+
urls = plan.get("eligible") if isinstance(plan.get("eligible"), list) else []
|
|
104
|
+
if not urls:
|
|
105
|
+
return {"schemaVersion": SCHEMA, "status": "no-op", "submitted": 0, "accepted": 0, "retryable": False, "mutation": True}
|
|
106
|
+
payload = {"host": urlparse(str(plan["origin"])).netloc, "key": str(plan["key"]),
|
|
107
|
+
"keyLocation": str(plan["keyLocation"]), "urlList": urls}
|
|
108
|
+
status_code: int | None = None
|
|
109
|
+
failure = None
|
|
110
|
+
try:
|
|
111
|
+
request = Request(endpoint, data=json.dumps(payload).encode("utf-8"), headers={"Content-Type": "application/json; charset=utf-8"}, method="POST")
|
|
112
|
+
with urlopen(request, timeout=30) as response:
|
|
113
|
+
status_code = int(response.status)
|
|
114
|
+
except HTTPError as error:
|
|
115
|
+
status_code = int(error.code)
|
|
116
|
+
failure = "provider returned HTTP status"
|
|
117
|
+
except (URLError, TimeoutError, OSError):
|
|
118
|
+
failure = "provider request failed"
|
|
119
|
+
accepted = status_code in ACCEPTED
|
|
120
|
+
retryable = status_code in {403, 429} or (status_code is not None and status_code >= 500) or failure is not None
|
|
121
|
+
entries = _state_entries(state)
|
|
122
|
+
if accepted:
|
|
123
|
+
stamp = _iso(_now())
|
|
124
|
+
entries.extend({"url": url, "statusCode": status_code, "acceptedAt": stamp, "batchId": hashlib.sha256((stamp + url).encode()).hexdigest()[:12]} for url in urls)
|
|
125
|
+
state.update({"schemaVersion": SCHEMA, "submissions": entries})
|
|
126
|
+
return {"schemaVersion": SCHEMA, "status": "accepted" if accepted else "retryable-failure" if retryable else "failed",
|
|
127
|
+
"submitted": len(urls), "accepted": len(urls) if accepted else 0, "statusCode": status_code,
|
|
128
|
+
"retryable": retryable, "failure": failure, "mutation": True}
|
|
@@ -14,6 +14,7 @@ MEDIA_SCHEMA = "maggie-media-uniqueness.v1"
|
|
|
14
14
|
INVENTORY_SCHEMA = "maggie-page-inventory.v1"
|
|
15
15
|
BINDING_SCHEMA = "maggie-section-bindings.v1"
|
|
16
16
|
IDEMPOTENCY_SCHEMA = "maggie-reconcile-contract.v1"
|
|
17
|
+
PAGE_KINDS = {"service", "variant", "category-hub", "ordinary", "blog-topic", "blog-archive", "blog-article", "homepage", "service-index", "service-category", "redirect", "asset"}
|
|
17
18
|
|
|
18
19
|
|
|
19
20
|
def _items(value: object, *keys: str) -> list[dict[str, Any]]:
|
|
@@ -116,8 +117,10 @@ def validate_media_uniqueness(value: object, *, across_siblings: bool = False) -
|
|
|
116
117
|
return {"schemaVersion": MEDIA_SCHEMA, "passed": not errors, "duplicates": errors, "pages": len(pages), "acrossSiblings": across_siblings}
|
|
117
118
|
|
|
118
119
|
|
|
119
|
-
def classify_inventory(value: object) -> dict[str, Any]:
|
|
120
|
+
def classify_inventory(value: object, *, require_page_kinds: bool = False, require_source_coverage: bool = False) -> dict[str, Any]:
|
|
120
121
|
pages = _items(value, "pages", "items", "routes")
|
|
122
|
+
if isinstance(value, dict):
|
|
123
|
+
pages += _items(value, "codeRenderedPages", "codeRenderedRoutes")
|
|
121
124
|
allowed = {"band", "code-rendered", "redirect", "asset", "unknown"}
|
|
122
125
|
records: list[dict[str, Any]] = []
|
|
123
126
|
errors: list[str] = []
|
|
@@ -127,7 +130,7 @@ def classify_inventory(value: object) -> dict[str, Any]:
|
|
|
127
130
|
if path in seen:
|
|
128
131
|
errors.append(f"duplicate inventory path: {path}")
|
|
129
132
|
seen.add(path)
|
|
130
|
-
explicit = str(page.get("kind") or page.get("classification") or "").casefold()
|
|
133
|
+
explicit = str(page.get("rendererKind") or page.get("rendering") or page.get("kind") or page.get("classification") or "").casefold()
|
|
131
134
|
if explicit in {"code", "code-rendered", "component", "source"}:
|
|
132
135
|
kind = "code-rendered"
|
|
133
136
|
elif explicit in {"band", "section", "sections"}:
|
|
@@ -140,11 +143,25 @@ def classify_inventory(value: object) -> dict[str, Any]:
|
|
|
140
143
|
kind = "code-rendered"
|
|
141
144
|
else:
|
|
142
145
|
kind = "unknown"
|
|
143
|
-
|
|
146
|
+
raw_page_kind = page.get("pageKind") or page.get("pageType")
|
|
147
|
+
if not raw_page_kind and str(page.get("kind") or "").casefold() in PAGE_KINDS:
|
|
148
|
+
raw_page_kind = page.get("kind")
|
|
149
|
+
page_kind = str(raw_page_kind or "").casefold().replace("_", "-")
|
|
150
|
+
records.append({"path": path, "kind": kind, "pageKind": page_kind or None})
|
|
144
151
|
if kind not in allowed or kind == "unknown":
|
|
145
152
|
errors.append(f"{path}: cannot classify published page")
|
|
153
|
+
if page_kind and page_kind not in PAGE_KINDS:
|
|
154
|
+
errors.append(f"{path}: unknown pageKind {page_kind}")
|
|
155
|
+
if require_page_kinds and not page_kind:
|
|
156
|
+
errors.append(f"{path}: pageKind is required")
|
|
146
157
|
counts = dict(sorted(Counter(item["kind"] for item in records).items()))
|
|
147
|
-
|
|
158
|
+
page_kind_counts = dict(sorted(Counter(item["pageKind"] for item in records if item["pageKind"]).items()))
|
|
159
|
+
coverage = value.get("sourceCoverage") if isinstance(value, dict) else None
|
|
160
|
+
if not isinstance(coverage, dict):
|
|
161
|
+
coverage = {"status": "unverified", "complete": False, "sources": []}
|
|
162
|
+
if require_source_coverage and coverage.get("complete") is not True:
|
|
163
|
+
errors.append("sourceCoverage.complete must be true when source coverage is required")
|
|
164
|
+
return {"schemaVersion": INVENTORY_SCHEMA, "passed": not errors and len(records) == len(seen), "records": records, "counts": counts, "pageKindCounts": page_kind_counts, "pageKinds": sorted(PAGE_KINDS), "bandCount": counts.get("band", 0), "codeRenderedCount": counts.get("code-rendered", 0), "errors": errors, "disjoint": True, "sourceCoverage": coverage}
|
|
148
165
|
|
|
149
166
|
|
|
150
167
|
def _reference_values(references: object) -> dict[str, set[str]]:
|
|
@@ -15,6 +15,8 @@ HARD_BYTES_LIMIT = 52_428_800
|
|
|
15
15
|
DEFAULT_CHUNK_TARGET = 500
|
|
16
16
|
EXCLUDED_PATHS = re.compile(r"/(search|find|login|draft|preview)(/|$)", re.I)
|
|
17
17
|
W3C_UTC_DATETIME = re.compile(r"^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}Z$")
|
|
18
|
+
FRESHNESS_MIN_SAMPLE = 10
|
|
19
|
+
FRESHNESS_CONCENTRATION_THRESHOLD = 0.75
|
|
18
20
|
|
|
19
21
|
|
|
20
22
|
def absolute_url(url: str, origin: str) -> bool:
|
|
@@ -100,6 +102,51 @@ def xml_file(urls: list[dict[str, str]]) -> str:
|
|
|
100
102
|
return '<?xml version="1.0" encoding="UTF-8"?>\n<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9" xmlns:xhtml="http://www.w3.org/1999/xhtml">\n' + body + ('\n' if body else '') + '</urlset>\n'
|
|
101
103
|
|
|
102
104
|
|
|
105
|
+
def freshness_report(routes: list[dict[str, str]], today: str | None = None) -> dict[str, object]:
|
|
106
|
+
"""Detect a suspiciously uniform lastmod distribution.
|
|
107
|
+
|
|
108
|
+
A structurally valid sitemap can still lose crawler trust when a migration
|
|
109
|
+
stamps nearly every URL with the same operational date. This is a warning
|
|
110
|
+
by default so unknown or intentionally batched content is not silently
|
|
111
|
+
rewritten; strict semantic validation promotes it to an error.
|
|
112
|
+
"""
|
|
113
|
+
dates: list[str] = []
|
|
114
|
+
for route in routes:
|
|
115
|
+
value = route.get("lastmod")
|
|
116
|
+
if not value:
|
|
117
|
+
continue
|
|
118
|
+
normalised = conventional_lastmod(str(value))
|
|
119
|
+
if W3C_UTC_DATETIME.fullmatch(normalised):
|
|
120
|
+
dates.append(normalised[:10])
|
|
121
|
+
counts: dict[str, int] = {}
|
|
122
|
+
for date in dates:
|
|
123
|
+
counts[date] = counts.get(date, 0) + 1
|
|
124
|
+
dominant_date, dominant_count = (max(counts.items(), key=lambda item: (item[1], item[0]))
|
|
125
|
+
if counts else (None, 0))
|
|
126
|
+
sample = len(dates)
|
|
127
|
+
ratio = dominant_count / sample if sample else 0.0
|
|
128
|
+
current = today or datetime.now(timezone.utc).date().isoformat()
|
|
129
|
+
suspicious = bool(sample >= FRESHNESS_MIN_SAMPLE and ratio >= FRESHNESS_CONCENTRATION_THRESHOLD)
|
|
130
|
+
warnings: list[str] = []
|
|
131
|
+
if suspicious:
|
|
132
|
+
date_label = "today's date" if dominant_date == current else str(dominant_date)
|
|
133
|
+
warnings.append(
|
|
134
|
+
f"lastmod distribution is concentrated on {date_label} "
|
|
135
|
+
f"({dominant_count}/{sample}, {ratio:.0%}); verify content-change provenance"
|
|
136
|
+
)
|
|
137
|
+
return {
|
|
138
|
+
"status": "warn" if warnings else "pass",
|
|
139
|
+
"datedRoutes": sample,
|
|
140
|
+
"distinctDates": len(counts),
|
|
141
|
+
"dominantDate": dominant_date,
|
|
142
|
+
"dominantCount": dominant_count,
|
|
143
|
+
"dominantRatio": round(ratio, 4),
|
|
144
|
+
"today": current,
|
|
145
|
+
"todayCount": counts.get(current, 0),
|
|
146
|
+
"warnings": warnings,
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
|
|
103
150
|
def build_plan(routes: list[dict[str, str]], origin: str, content_types: set[str], chunk_target: int = DEFAULT_CHUNK_TARGET, previous: dict | None = None) -> dict:
|
|
104
151
|
if chunk_target < 1 or chunk_target > HARD_URL_LIMIT:
|
|
105
152
|
raise ValueError("chunk target must be between 1 and 50000")
|
|
@@ -134,6 +181,7 @@ def validate_plan_data(plan: dict, strict_semantic: bool = False) -> dict:
|
|
|
134
181
|
if not urlparse(origin).scheme or not urlparse(origin).netloc:
|
|
135
182
|
errors.append("origin must be an absolute HTTP(S) URL")
|
|
136
183
|
expected_urls = []
|
|
184
|
+
all_routes: list[dict[str, str]] = []
|
|
137
185
|
for group in plan.get("groups", []):
|
|
138
186
|
for chunk in group.get("chunks", []):
|
|
139
187
|
expected_urls.append(chunk.get("url"))
|
|
@@ -142,6 +190,7 @@ def validate_plan_data(plan: dict, strict_semantic: bool = False) -> dict:
|
|
|
142
190
|
if chunk.get("bytes", 0) > HARD_BYTES_LIMIT:
|
|
143
191
|
errors.append(f"chunk exceeds byte limit: {chunk.get('filename')}")
|
|
144
192
|
for route in chunk.get("routes", []):
|
|
193
|
+
all_routes.append(route)
|
|
145
194
|
if route.get("indexable") is False or route.get("searchable") is False:
|
|
146
195
|
errors.append(f"non-indexable/searchable route was emitted: {route.get('url')}")
|
|
147
196
|
if route.get("canonicalUrl") and route.get("canonicalUrl") != route.get("url"):
|
|
@@ -187,6 +236,8 @@ def validate_plan_data(plan: dict, strict_semantic: bool = False) -> dict:
|
|
|
187
236
|
errors.append("sitemap index is not valid XML")
|
|
188
237
|
if not expected_urls:
|
|
189
238
|
warnings.append("no sitemap chunks generated; empty content types are not advertised")
|
|
239
|
+
freshness = freshness_report(all_routes)
|
|
240
|
+
warnings.extend(freshness["warnings"])
|
|
190
241
|
if strict_semantic:
|
|
191
242
|
for group in plan.get("groups", []):
|
|
192
243
|
for chunk in group.get("chunks", []):
|
|
@@ -195,7 +246,7 @@ def validate_plan_data(plan: dict, strict_semantic: bool = False) -> dict:
|
|
|
195
246
|
if "contentWords" not in route: warnings.append(f"semantic content evidence missing: {route.get('url')}")
|
|
196
247
|
if warnings:
|
|
197
248
|
errors.extend(warnings)
|
|
198
|
-
return {"status": "pass" if not errors else "fail", "errors": errors, "warnings": warnings}
|
|
249
|
+
return {"status": "pass" if not errors else "fail", "errors": errors, "warnings": warnings, "freshness": freshness}
|
|
199
250
|
|
|
200
251
|
|
|
201
252
|
def agent_files(routes: list[dict[str, str]], origin: str, locale: str = "en") -> dict[str, str]:
|