@topy-ai/maggie 0.7.13 → 0.7.15

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,128 @@
1
+ """Provider-neutral, change-driven IndexNow planning and submission state."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import hashlib
6
+ import json
7
+ from datetime import datetime, timedelta, timezone
8
+ from pathlib import Path
9
+ from typing import Any
10
+ from urllib.error import HTTPError, URLError
11
+ from urllib.parse import urlparse
12
+ from urllib.request import Request, urlopen
13
+
14
+
15
+ SCHEMA = "maggie-indexnow.v1"
16
+ DEFAULT_GUARD_HOURS = 24
17
+ ACCEPTED = {200, 202}
18
+
19
+
20
+ def _now(value: str | None = None) -> datetime:
21
+ if value:
22
+ return datetime.fromisoformat(value.replace("Z", "+00:00")).astimezone(timezone.utc)
23
+ return datetime.now(timezone.utc)
24
+
25
+
26
+ def _iso(value: datetime) -> str:
27
+ return value.astimezone(timezone.utc).isoformat(timespec="seconds").replace("+00:00", "Z")
28
+
29
+
30
+ def normalise_urls(origin: str, urls: list[object]) -> tuple[list[str], list[str]]:
31
+ """Return valid same-origin URLs and explicit rejects."""
32
+ base = origin.rstrip("/")
33
+ origin_host = urlparse(base).netloc
34
+ accepted: set[str] = set()
35
+ rejected: list[str] = []
36
+ for raw in urls:
37
+ value = str(raw or "").strip()
38
+ if not value:
39
+ continue
40
+ candidate = value if value.startswith(("http://", "https://")) else f"{base}/{value.lstrip('/')}"
41
+ parsed = urlparse(candidate)
42
+ if parsed.scheme not in {"http", "https"} or parsed.netloc != origin_host or parsed.fragment:
43
+ rejected.append(value)
44
+ continue
45
+ accepted.add(candidate)
46
+ return sorted(accepted), rejected
47
+
48
+
49
+ def _state_entries(state: object) -> list[dict[str, Any]]:
50
+ if not isinstance(state, dict) or not isinstance(state.get("submissions"), list):
51
+ return []
52
+ return [item for item in state["submissions"] if isinstance(item, dict)]
53
+
54
+
55
+ def plan_indexnow(origin: str, urls: list[object], state: object | None = None,
56
+ *, key: str, key_location: str | None = None,
57
+ guard_hours: int = DEFAULT_GUARD_HOURS, now: str | None = None) -> dict[str, Any]:
58
+ if not key.strip():
59
+ raise ValueError("IndexNow key is required; it is a public ownership key, not a provider secret")
60
+ eligible, rejected = normalise_urls(origin, urls)
61
+ cutoff = _now(now) - timedelta(hours=guard_hours)
62
+ accepted_at: dict[str, datetime] = {}
63
+ for entry in _state_entries(state):
64
+ if entry.get("statusCode") not in ACCEPTED or not entry.get("acceptedAt"):
65
+ continue
66
+ try:
67
+ stamp = _now(str(entry["acceptedAt"]))
68
+ except ValueError:
69
+ continue
70
+ if stamp >= cutoff:
71
+ accepted_at[str(entry.get("url"))] = stamp
72
+ held_back = [{"url": url, "acceptedAt": _iso(accepted_at[url]), "reason": f"accepted within {guard_hours} hours"}
73
+ for url in eligible if url in accepted_at]
74
+ held_set = {item["url"] for item in held_back}
75
+ return {
76
+ "schemaVersion": SCHEMA,
77
+ "origin": origin.rstrip("/"),
78
+ "key": key.strip(),
79
+ "keyLocation": key_location or f"{origin.rstrip('/')}/{key.strip()}.txt",
80
+ "guardHours": guard_hours,
81
+ "eligible": [url for url in eligible if url not in held_set],
82
+ "heldBack": held_back,
83
+ "rejected": rejected,
84
+ "verification": {"required": True, "status": "not-checked"},
85
+ "plannedAt": _iso(_now(now)),
86
+ "mutation": False,
87
+ }
88
+
89
+
90
+ def check_key_file(public_dir: Path, key_file: str, key: str) -> dict[str, Any]:
91
+ if not key.strip():
92
+ raise ValueError("key is required")
93
+ relative = Path(key_file)
94
+ if relative.is_absolute() or ".." in relative.parts or len(relative.parts) != 1:
95
+ raise ValueError("key file must be a single filename inside public-dir")
96
+ target = public_dir / relative
97
+ served = target.read_text(encoding="utf-8").strip() if target.exists() else None
98
+ return {"schemaVersion": SCHEMA, "path": str(relative), "exists": target.exists(),
99
+ "matches": served == key.strip(), "status": "pass" if served == key.strip() else "fail"}
100
+
101
+
102
+ def submit_plan(plan: dict[str, Any], state: dict[str, Any], endpoint: str) -> dict[str, Any]:
103
+ urls = plan.get("eligible") if isinstance(plan.get("eligible"), list) else []
104
+ if not urls:
105
+ return {"schemaVersion": SCHEMA, "status": "no-op", "submitted": 0, "accepted": 0, "retryable": False, "mutation": True}
106
+ payload = {"host": urlparse(str(plan["origin"])).netloc, "key": str(plan["key"]),
107
+ "keyLocation": str(plan["keyLocation"]), "urlList": urls}
108
+ status_code: int | None = None
109
+ failure = None
110
+ try:
111
+ request = Request(endpoint, data=json.dumps(payload).encode("utf-8"), headers={"Content-Type": "application/json; charset=utf-8"}, method="POST")
112
+ with urlopen(request, timeout=30) as response:
113
+ status_code = int(response.status)
114
+ except HTTPError as error:
115
+ status_code = int(error.code)
116
+ failure = "provider returned HTTP status"
117
+ except (URLError, TimeoutError, OSError):
118
+ failure = "provider request failed"
119
+ accepted = status_code in ACCEPTED
120
+ retryable = status_code in {403, 429} or (status_code is not None and status_code >= 500) or failure is not None
121
+ entries = _state_entries(state)
122
+ if accepted:
123
+ stamp = _iso(_now())
124
+ entries.extend({"url": url, "statusCode": status_code, "acceptedAt": stamp, "batchId": hashlib.sha256((stamp + url).encode()).hexdigest()[:12]} for url in urls)
125
+ state.update({"schemaVersion": SCHEMA, "submissions": entries})
126
+ return {"schemaVersion": SCHEMA, "status": "accepted" if accepted else "retryable-failure" if retryable else "failed",
127
+ "submitted": len(urls), "accepted": len(urls) if accepted else 0, "statusCode": status_code,
128
+ "retryable": retryable, "failure": failure, "mutation": True}
@@ -14,6 +14,7 @@ MEDIA_SCHEMA = "maggie-media-uniqueness.v1"
14
14
  INVENTORY_SCHEMA = "maggie-page-inventory.v1"
15
15
  BINDING_SCHEMA = "maggie-section-bindings.v1"
16
16
  IDEMPOTENCY_SCHEMA = "maggie-reconcile-contract.v1"
17
+ PAGE_KINDS = {"service", "variant", "category-hub", "ordinary", "blog-topic", "blog-archive", "blog-article", "homepage", "service-index", "service-category", "redirect", "asset"}
17
18
 
18
19
 
19
20
  def _items(value: object, *keys: str) -> list[dict[str, Any]]:
@@ -116,8 +117,10 @@ def validate_media_uniqueness(value: object, *, across_siblings: bool = False) -
116
117
  return {"schemaVersion": MEDIA_SCHEMA, "passed": not errors, "duplicates": errors, "pages": len(pages), "acrossSiblings": across_siblings}
117
118
 
118
119
 
119
- def classify_inventory(value: object) -> dict[str, Any]:
120
+ def classify_inventory(value: object, *, require_page_kinds: bool = False, require_source_coverage: bool = False) -> dict[str, Any]:
120
121
  pages = _items(value, "pages", "items", "routes")
122
+ if isinstance(value, dict):
123
+ pages += _items(value, "codeRenderedPages", "codeRenderedRoutes")
121
124
  allowed = {"band", "code-rendered", "redirect", "asset", "unknown"}
122
125
  records: list[dict[str, Any]] = []
123
126
  errors: list[str] = []
@@ -127,7 +130,7 @@ def classify_inventory(value: object) -> dict[str, Any]:
127
130
  if path in seen:
128
131
  errors.append(f"duplicate inventory path: {path}")
129
132
  seen.add(path)
130
- explicit = str(page.get("kind") or page.get("classification") or "").casefold()
133
+ explicit = str(page.get("rendererKind") or page.get("rendering") or page.get("kind") or page.get("classification") or "").casefold()
131
134
  if explicit in {"code", "code-rendered", "component", "source"}:
132
135
  kind = "code-rendered"
133
136
  elif explicit in {"band", "section", "sections"}:
@@ -140,11 +143,25 @@ def classify_inventory(value: object) -> dict[str, Any]:
140
143
  kind = "code-rendered"
141
144
  else:
142
145
  kind = "unknown"
143
- records.append({"path": path, "kind": kind})
146
+ raw_page_kind = page.get("pageKind") or page.get("pageType")
147
+ if not raw_page_kind and str(page.get("kind") or "").casefold() in PAGE_KINDS:
148
+ raw_page_kind = page.get("kind")
149
+ page_kind = str(raw_page_kind or "").casefold().replace("_", "-")
150
+ records.append({"path": path, "kind": kind, "pageKind": page_kind or None})
144
151
  if kind not in allowed or kind == "unknown":
145
152
  errors.append(f"{path}: cannot classify published page")
153
+ if page_kind and page_kind not in PAGE_KINDS:
154
+ errors.append(f"{path}: unknown pageKind {page_kind}")
155
+ if require_page_kinds and not page_kind:
156
+ errors.append(f"{path}: pageKind is required")
146
157
  counts = dict(sorted(Counter(item["kind"] for item in records).items()))
147
- return {"schemaVersion": INVENTORY_SCHEMA, "passed": not errors and len(records) == len(seen), "records": records, "counts": counts, "bandCount": counts.get("band", 0), "codeRenderedCount": counts.get("code-rendered", 0), "errors": errors, "disjoint": True}
158
+ page_kind_counts = dict(sorted(Counter(item["pageKind"] for item in records if item["pageKind"]).items()))
159
+ coverage = value.get("sourceCoverage") if isinstance(value, dict) else None
160
+ if not isinstance(coverage, dict):
161
+ coverage = {"status": "unverified", "complete": False, "sources": []}
162
+ if require_source_coverage and coverage.get("complete") is not True:
163
+ errors.append("sourceCoverage.complete must be true when source coverage is required")
164
+ return {"schemaVersion": INVENTORY_SCHEMA, "passed": not errors and len(records) == len(seen), "records": records, "counts": counts, "pageKindCounts": page_kind_counts, "pageKinds": sorted(PAGE_KINDS), "bandCount": counts.get("band", 0), "codeRenderedCount": counts.get("code-rendered", 0), "errors": errors, "disjoint": True, "sourceCoverage": coverage}
148
165
 
149
166
 
150
167
  def _reference_values(references: object) -> dict[str, set[str]]:
@@ -15,6 +15,8 @@ HARD_BYTES_LIMIT = 52_428_800
15
15
  DEFAULT_CHUNK_TARGET = 500
16
16
  EXCLUDED_PATHS = re.compile(r"/(search|find|login|draft|preview)(/|$)", re.I)
17
17
  W3C_UTC_DATETIME = re.compile(r"^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}Z$")
18
+ FRESHNESS_MIN_SAMPLE = 10
19
+ FRESHNESS_CONCENTRATION_THRESHOLD = 0.75
18
20
 
19
21
 
20
22
  def absolute_url(url: str, origin: str) -> bool:
@@ -100,6 +102,51 @@ def xml_file(urls: list[dict[str, str]]) -> str:
100
102
  return '<?xml version="1.0" encoding="UTF-8"?>\n<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9" xmlns:xhtml="http://www.w3.org/1999/xhtml">\n' + body + ('\n' if body else '') + '</urlset>\n'
101
103
 
102
104
 
105
+ def freshness_report(routes: list[dict[str, str]], today: str | None = None) -> dict[str, object]:
106
+ """Detect a suspiciously uniform lastmod distribution.
107
+
108
+ A structurally valid sitemap can still lose crawler trust when a migration
109
+ stamps nearly every URL with the same operational date. This is a warning
110
+ by default so unknown or intentionally batched content is not silently
111
+ rewritten; strict semantic validation promotes it to an error.
112
+ """
113
+ dates: list[str] = []
114
+ for route in routes:
115
+ value = route.get("lastmod")
116
+ if not value:
117
+ continue
118
+ normalised = conventional_lastmod(str(value))
119
+ if W3C_UTC_DATETIME.fullmatch(normalised):
120
+ dates.append(normalised[:10])
121
+ counts: dict[str, int] = {}
122
+ for date in dates:
123
+ counts[date] = counts.get(date, 0) + 1
124
+ dominant_date, dominant_count = (max(counts.items(), key=lambda item: (item[1], item[0]))
125
+ if counts else (None, 0))
126
+ sample = len(dates)
127
+ ratio = dominant_count / sample if sample else 0.0
128
+ current = today or datetime.now(timezone.utc).date().isoformat()
129
+ suspicious = bool(sample >= FRESHNESS_MIN_SAMPLE and ratio >= FRESHNESS_CONCENTRATION_THRESHOLD)
130
+ warnings: list[str] = []
131
+ if suspicious:
132
+ date_label = "today's date" if dominant_date == current else str(dominant_date)
133
+ warnings.append(
134
+ f"lastmod distribution is concentrated on {date_label} "
135
+ f"({dominant_count}/{sample}, {ratio:.0%}); verify content-change provenance"
136
+ )
137
+ return {
138
+ "status": "warn" if warnings else "pass",
139
+ "datedRoutes": sample,
140
+ "distinctDates": len(counts),
141
+ "dominantDate": dominant_date,
142
+ "dominantCount": dominant_count,
143
+ "dominantRatio": round(ratio, 4),
144
+ "today": current,
145
+ "todayCount": counts.get(current, 0),
146
+ "warnings": warnings,
147
+ }
148
+
149
+
103
150
  def build_plan(routes: list[dict[str, str]], origin: str, content_types: set[str], chunk_target: int = DEFAULT_CHUNK_TARGET, previous: dict | None = None) -> dict:
104
151
  if chunk_target < 1 or chunk_target > HARD_URL_LIMIT:
105
152
  raise ValueError("chunk target must be between 1 and 50000")
@@ -134,6 +181,7 @@ def validate_plan_data(plan: dict, strict_semantic: bool = False) -> dict:
134
181
  if not urlparse(origin).scheme or not urlparse(origin).netloc:
135
182
  errors.append("origin must be an absolute HTTP(S) URL")
136
183
  expected_urls = []
184
+ all_routes: list[dict[str, str]] = []
137
185
  for group in plan.get("groups", []):
138
186
  for chunk in group.get("chunks", []):
139
187
  expected_urls.append(chunk.get("url"))
@@ -142,6 +190,7 @@ def validate_plan_data(plan: dict, strict_semantic: bool = False) -> dict:
142
190
  if chunk.get("bytes", 0) > HARD_BYTES_LIMIT:
143
191
  errors.append(f"chunk exceeds byte limit: {chunk.get('filename')}")
144
192
  for route in chunk.get("routes", []):
193
+ all_routes.append(route)
145
194
  if route.get("indexable") is False or route.get("searchable") is False:
146
195
  errors.append(f"non-indexable/searchable route was emitted: {route.get('url')}")
147
196
  if route.get("canonicalUrl") and route.get("canonicalUrl") != route.get("url"):
@@ -187,6 +236,8 @@ def validate_plan_data(plan: dict, strict_semantic: bool = False) -> dict:
187
236
  errors.append("sitemap index is not valid XML")
188
237
  if not expected_urls:
189
238
  warnings.append("no sitemap chunks generated; empty content types are not advertised")
239
+ freshness = freshness_report(all_routes)
240
+ warnings.extend(freshness["warnings"])
190
241
  if strict_semantic:
191
242
  for group in plan.get("groups", []):
192
243
  for chunk in group.get("chunks", []):
@@ -195,7 +246,7 @@ def validate_plan_data(plan: dict, strict_semantic: bool = False) -> dict:
195
246
  if "contentWords" not in route: warnings.append(f"semantic content evidence missing: {route.get('url')}")
196
247
  if warnings:
197
248
  errors.extend(warnings)
198
- return {"status": "pass" if not errors else "fail", "errors": errors, "warnings": warnings}
249
+ return {"status": "pass" if not errors else "fail", "errors": errors, "warnings": warnings, "freshness": freshness}
199
250
 
200
251
 
201
252
  def agent_files(routes: list[dict[str, str]], origin: str, locale: str = "en") -> dict[str, str]:
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@topy-ai/maggie",
3
- "version": "0.7.13",
3
+ "version": "0.7.15",
4
4
  "description": "Install and manage Maggie Skills for AI coding agents",
5
5
  "license": "MIT",
6
6
  "type": "module",