@topy-ai/maggie 0.7.11 → 0.7.12
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md
CHANGED
|
@@ -218,8 +218,8 @@ artifact schemas.
|
|
|
218
218
|
Recommended upgrade sequence for the current release:
|
|
219
219
|
|
|
220
220
|
```bash
|
|
221
|
-
npx @topy-ai/maggie@0.7.
|
|
222
|
-
npx @topy-ai/maggie@0.7.
|
|
221
|
+
npx @topy-ai/maggie@0.7.12 update --project . --force
|
|
222
|
+
npx @topy-ai/maggie@0.7.12 cleanup --project .
|
|
223
223
|
```
|
|
224
224
|
|
|
225
225
|
Maintainers should pass npm credentials through the repository helper, never
|
|
@@ -229,7 +229,7 @@ as a command-line argument:
|
|
|
229
229
|
node scripts/publish-npm.mjs --maggie-env-file ../.env
|
|
230
230
|
```
|
|
231
231
|
|
|
232
|
-
The 0.7.
|
|
232
|
+
The 0.7.12 workflow adds served-content equivalence checks, query-route
|
|
233
233
|
baseline exclusions, changed-surface render evidence gates, generated skill
|
|
234
234
|
catalogs, component-binding audits, and icon-family noise filtering. It also
|
|
235
235
|
includes the 0.7.9 nested section-field contracts, renderer-backed examples,
|
|
@@ -255,7 +255,9 @@ also adds image-aware content equivalence with explicit legacy-baseline
|
|
|
255
255
|
limitations, locale checks for stored image alt text, a validated label/value
|
|
256
256
|
`pairs` section, registry fan-out checks, idempotent reconciliation, per-step
|
|
257
257
|
and section-scoped design evidence, catch-all source confidence, shell-safe VPS
|
|
258
|
-
SSH guidance,
|
|
258
|
+
SSH guidance, compatible feedback CLI examples, conventional sitemap
|
|
259
|
+
formatting, response-header guidance, and access-log diagnostics for absent
|
|
260
|
+
crawler requests. The release
|
|
259
261
|
retains the existing service matching,
|
|
260
262
|
localization, seed-manifest, lockfile/analytics, sitemap, deployment and
|
|
261
263
|
rollback workflows.
|
package/bin/maggie.js
CHANGED
|
@@ -112,7 +112,7 @@ Usage:
|
|
|
112
112
|
maggie localization <extract|plan|generate|preview|validate|review|publish|stale|glossary> [options]
|
|
113
113
|
maggie seo performance|images|sitemap [options] (sitemap supports strict validate and agent-files)
|
|
114
114
|
maggie feedback <collect|preview|submit|list> [options]
|
|
115
|
-
maggie site-audit URL [--crawl] [--languages en-GB,es-MX,ja-JP] [--check-hreflang]
|
|
115
|
+
maggie site-audit URL [--crawl] [--access-log FILE] [--require-sitemap-request] [--languages en-GB,es-MX,ja-JP] [--check-hreflang]
|
|
116
116
|
maggie site-audit URL --crawl --save-baseline FILE --reviewer NAME
|
|
117
117
|
maggie site-audit URL --crawl --baseline FILE
|
|
118
118
|
maggie browser-audit URL --browse PATH --output DIR --required SELECTOR [--sticky SELECTOR]
|
|
@@ -59,6 +59,26 @@ If the change date is unknown, omit `lastmod`. An empty content-type does not
|
|
|
59
59
|
need a sitemap chunk in the sitemap index; serving an empty endpoint and
|
|
60
60
|
advertising it are separate decisions.
|
|
61
61
|
|
|
62
|
+
Generated sitemap XML uses the conventional readable shape by default: one
|
|
63
|
+
`<url>`/`<sitemap>` entry per block, UTF-8 XML, and date-only `lastmod` evidence
|
|
64
|
+
rendered as a full UTC W3C datetime. The plan also exposes the response
|
|
65
|
+
contract (`text/xml; charset=utf-8` and `X-Robots-Tag: all`) for the framework
|
|
66
|
+
route or reverse proxy to apply; writing a file cannot set HTTP headers itself.
|
|
67
|
+
When Search Console reports a valid sitemap as unfetched, probe success alone
|
|
68
|
+
is not a diagnosis. Supply a sanitized local access log to distinguish
|
|
69
|
+
`googlebot-request-observed`, another-client-only traffic, and
|
|
70
|
+
`no-matching-request`:
|
|
71
|
+
|
|
72
|
+
```bash
|
|
73
|
+
maggie site-audit https://example.com --access-log /var/log/nginx/access.log
|
|
74
|
+
maggie site-audit https://example.com --access-log /var/log/nginx/access.log \
|
|
75
|
+
--require-sitemap-request
|
|
76
|
+
```
|
|
77
|
+
|
|
78
|
+
The strict flag is for a review window where a crawler request is expected; it
|
|
79
|
+
fails only when the supplied log contains no sitemap request. Never upload raw
|
|
80
|
+
logs or include IPs, credentials, or query data in feedback or reports.
|
|
81
|
+
|
|
62
82
|
## Freeze and compare a reviewed site
|
|
63
83
|
|
|
64
84
|
```bash
|
|
@@ -257,6 +257,15 @@ def sitemap_urls(base: str) -> tuple[list[str], list[dict[str, str]]]:
|
|
|
257
257
|
return list(dict.fromkeys(pages)), violations
|
|
258
258
|
|
|
259
259
|
|
|
260
|
+
def access_log_sitemap_check(path: Path, sitemap_path: str = "/sitemap.xml") -> dict[str, object]:
|
|
261
|
+
"""Distinguish a sitemap fetch failure from a crawler that never asked."""
|
|
262
|
+
text = path.read_text(encoding="utf-8", errors="replace")
|
|
263
|
+
escaped = re.escape(sitemap_path)
|
|
264
|
+
requests = [line for line in text.splitlines() if re.search(rf"(?:GET|HEAD)\s+{escaped}(?:[?\s\"]|$)", line, re.I)]
|
|
265
|
+
googlebot = [line for line in requests if "googlebot" in line.lower()]
|
|
266
|
+
return {"provided": True, "path": str(path), "requestCount": len(requests), "googlebotRequestCount": len(googlebot), "status": "requested" if requests else "not-requested", "diagnosis": "googlebot-request-observed" if googlebot else ("other-client-request-only" if requests else "no-matching-request")}
|
|
267
|
+
|
|
268
|
+
|
|
260
269
|
def main() -> int:
|
|
261
270
|
parser = argparse.ArgumentParser()
|
|
262
271
|
parser.add_argument("url")
|
|
@@ -269,6 +278,8 @@ def main() -> int:
|
|
|
269
278
|
parser.add_argument("--markets", help="comma-separated markets: global,uk,us")
|
|
270
279
|
parser.add_argument("--check-hreflang", action="store_true")
|
|
271
280
|
parser.add_argument("--check-translation-completeness", action="store_true")
|
|
281
|
+
parser.add_argument("--access-log", type=Path, help="optional local access log for crawler-request diagnosis")
|
|
282
|
+
parser.add_argument("--require-sitemap-request", action="store_true", help="fail unless the supplied access log contains a sitemap request")
|
|
272
283
|
baseline_args = parser.add_mutually_exclusive_group()
|
|
273
284
|
baseline_args.add_argument("--save-baseline", type=Path, help="create a new reviewed contract from a passing complete crawl")
|
|
274
285
|
baseline_args.add_argument("--baseline", type=Path, help="fail on differences from a reviewed contract")
|
|
@@ -326,6 +337,12 @@ def main() -> int:
|
|
|
326
337
|
"mentions_sitemap": "sitemap" in body.lower() if name == "robots" else None,
|
|
327
338
|
"url_count": len(re.findall(r"<loc>.*?</loc>", body, re.I | re.S)) if name == "sitemap" else None,
|
|
328
339
|
}
|
|
340
|
+
if name == "sitemap" and args.access_log:
|
|
341
|
+
access = access_log_sitemap_check(args.access_log)
|
|
342
|
+
checks["sitemap_access_log"] = {"ok": not args.require_sitemap_request or access["requestCount"] > 0, **access}
|
|
343
|
+
checks[name]["access_log"] = access
|
|
344
|
+
if args.require_sitemap_request and not access["requestCount"]:
|
|
345
|
+
checks[name]["ok"] = False
|
|
329
346
|
except Exception as exc:
|
|
330
347
|
checks[name] = {"ok": False, "error": type(exc).__name__}
|
|
331
348
|
|
|
@@ -74,13 +74,29 @@ def filter_routes(routes: list[dict[str, str]], origin: str, content_types: set[
|
|
|
74
74
|
return groups, excluded
|
|
75
75
|
|
|
76
76
|
|
|
77
|
+
def conventional_lastmod(value: str) -> str:
|
|
78
|
+
"""Render date-only source evidence as a W3C/ISO UTC datetime."""
|
|
79
|
+
if re.fullmatch(r"\d{4}-\d{2}-\d{2}", value):
|
|
80
|
+
return value + "T00:00:00Z"
|
|
81
|
+
try:
|
|
82
|
+
parsed = datetime.fromisoformat(value.replace("Z", "+00:00"))
|
|
83
|
+
except ValueError:
|
|
84
|
+
return value
|
|
85
|
+
if parsed.tzinfo is None:
|
|
86
|
+
parsed = parsed.replace(tzinfo=timezone.utc)
|
|
87
|
+
return parsed.astimezone(timezone.utc).isoformat(timespec="seconds").replace("+00:00", "Z")
|
|
88
|
+
|
|
89
|
+
|
|
77
90
|
def xml_file(urls: list[dict[str, str]]) -> str:
|
|
78
91
|
rows = []
|
|
79
92
|
for item in urls:
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
93
|
+
fields = [f" <loc>{escape(item['url'])}</loc>"]
|
|
94
|
+
if item.get("lastmod"):
|
|
95
|
+
fields.append(f" <lastmod>{escape(conventional_lastmod(item['lastmod']))}</lastmod>")
|
|
96
|
+
fields.extend(f' <xhtml:link rel="alternate" hreflang="{escape(str(lang))}" href="{escape(str(url))}" />' for lang, url in sorted((item.get("alternates") or {}).items()))
|
|
97
|
+
rows.append(" <url>\n" + "\n".join(fields) + "\n </url>")
|
|
98
|
+
body = "\n".join(rows)
|
|
99
|
+
return '<?xml version="1.0" encoding="UTF-8"?>\n<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9" xmlns:xhtml="http://www.w3.org/1999/xhtml">\n' + body + ('\n' if body else '') + '</urlset>\n'
|
|
84
100
|
|
|
85
101
|
|
|
86
102
|
def build_plan(routes: list[dict[str, str]], origin: str, content_types: set[str], chunk_target: int = DEFAULT_CHUNK_TARGET, previous: dict | None = None) -> dict:
|
|
@@ -100,11 +116,12 @@ def build_plan(routes: list[dict[str, str]], origin: str, content_types: set[str
|
|
|
100
116
|
chunks.append({"filename": filename, "url": url, "entries": len(chunk_entries), "routes": chunk_entries, "bytes": len(xml.encode()), "sha256": hashlib.sha256(xml.encode()).hexdigest(), "xml": xml})
|
|
101
117
|
sitemap_urls.append(url)
|
|
102
118
|
group_plans.append({"contentType": content_type, "chunks": chunks})
|
|
103
|
-
|
|
119
|
+
index_rows = "\n".join(f" <sitemap>\n <loc>{escape(url)}</loc>\n </sitemap>" for url in sitemap_urls)
|
|
120
|
+
index_xml = '<?xml version="1.0" encoding="UTF-8"?>\n<sitemapindex xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">\n' + index_rows + ('\n' if index_rows else '') + '</sitemapindex>\n'
|
|
104
121
|
current = set(sitemap_urls)
|
|
105
122
|
old = set((previous or {}).get("sitemapUrls", []))
|
|
106
123
|
redirects = [{"from": url, "to": origin.rstrip("/") + "/sitemap.xml", "reason": "previously advertised sitemap removed"} for url in sorted(old - current)]
|
|
107
|
-
plan = {"schemaVersion": "maggie-seo-sitemap-plan.v1", "planId": "sitemap-" + datetime.now(timezone.utc).strftime("%Y%m%d%H%M%S"), "origin": origin.rstrip("/"), "groups": group_plans, "index": {"url": origin.rstrip("/") + "/sitemap.xml", "bytes": len(index_xml.encode()), "xml": index_xml, "sitemapUrls": sitemap_urls}, "redirects": redirects, "excluded": excluded}
|
|
124
|
+
plan = {"schemaVersion": "maggie-seo-sitemap-plan.v1", "planId": "sitemap-" + datetime.now(timezone.utc).strftime("%Y%m%d%H%M%S"), "origin": origin.rstrip("/"), "groups": group_plans, "index": {"url": origin.rstrip("/") + "/sitemap.xml", "bytes": len(index_xml.encode()), "xml": index_xml, "sitemapUrls": sitemap_urls, "responseHeaders": {"content-type": "text/xml; charset=utf-8", "x-robots-tag": "all"}}, "redirects": redirects, "excluded": excluded}
|
|
108
125
|
plan["validation"] = validate_plan_data(plan)
|
|
109
126
|
return plan
|
|
110
127
|
|