@topy-ai/maggie 0.5.0 → 0.5.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md
CHANGED
|
@@ -80,8 +80,8 @@ with `--confirm`, removes only known retired EmDash artifacts. It preserves
|
|
|
80
80
|
Recommended upgrade sequence for the current release:
|
|
81
81
|
|
|
82
82
|
```bash
|
|
83
|
-
npx @topy-ai/maggie@0.
|
|
84
|
-
npx @topy-ai/maggie@0.
|
|
83
|
+
npx @topy-ai/maggie@0.5.1 update --project . --force
|
|
84
|
+
npx @topy-ai/maggie@0.5.1 cleanup --project .
|
|
85
85
|
```
|
|
86
86
|
|
|
87
87
|
## MaggieDash lifecycle
|
package/bin/maggie.js
CHANGED
|
@@ -208,7 +208,11 @@ function install(args) {
|
|
|
208
208
|
const tools = join(root, "tools");
|
|
209
209
|
if (existsSync(TOOLS_ROOT)) {
|
|
210
210
|
for (const group of ["clis", "runtime", "integrations"]) {
|
|
211
|
-
if (existsSync(join(TOOLS_ROOT, group)))
|
|
211
|
+
if (existsSync(join(TOOLS_ROOT, group))) {
|
|
212
|
+
const target = join(tools, group);
|
|
213
|
+
if (existsSync(target)) syncTree(join(TOOLS_ROOT, group), target, force);
|
|
214
|
+
else copyIfMissing(join(TOOLS_ROOT, group), target, force);
|
|
215
|
+
}
|
|
212
216
|
}
|
|
213
217
|
}
|
|
214
218
|
if (existsSync(join(DESIGN_ROOT, "SPA-DESIGN.md"))) {
|
|
@@ -244,7 +248,10 @@ function update(args) {
|
|
|
244
248
|
if (existsSync(REFERENCES_ROOT) && existsSync(join(agentRoot, "references"))) updated += syncTree(REFERENCES_ROOT, join(agentRoot, "references"), force);
|
|
245
249
|
}
|
|
246
250
|
const tools = join(root, "tools");
|
|
247
|
-
if (existsSync(TOOLS_ROOT)) for (const group of ["clis", "integrations"]) if (existsSync(join(
|
|
251
|
+
if (existsSync(TOOLS_ROOT)) for (const group of ["clis", "runtime", "integrations"]) if (existsSync(join(TOOLS_ROOT, group))) {
|
|
252
|
+
const target = join(tools, group);
|
|
253
|
+
if (existsSync(target)) updated += syncTree(join(TOOLS_ROOT, group), target, force);
|
|
254
|
+
}
|
|
248
255
|
if (existsSync(join(DESIGN_ROOT, "SPA-DESIGN.md")) && existsSync(join(root, ".maggie", "design-reference"))) {
|
|
249
256
|
updated += syncTree(DESIGN_ROOT, join(root, ".maggie", "design-reference"), force);
|
|
250
257
|
}
|
|
@@ -30,7 +30,10 @@ interpretation, and content creation remain agent-reviewed decisions.
|
|
|
30
30
|
With `--crawl`, it resolves sitemap indexes and audits every listed HTML route
|
|
31
31
|
for HTTP 200, one H1, title/description, locale, canonical, Open Graph/Twitter
|
|
32
32
|
metadata, valid entity JSON-LD, and image alt text. Non-HTML endpoints such as
|
|
33
|
-
RSS must not be included in the HTML page sitemap.
|
|
33
|
+
RSS must not be included in the HTML page sitemap. Sitemap `<loc>` values must
|
|
34
|
+
be absolute HTTP(S) URLs; the audit reports relative values as a failure even
|
|
35
|
+
when it can resolve them for continued crawling, so one malformed entry cannot
|
|
36
|
+
hide a protocol violation.
|
|
34
37
|
Use `--output <project>/docs/site-audit.json` to persist the evidence used by
|
|
35
38
|
the release review.
|
|
36
39
|
|
|
@@ -134,10 +134,11 @@ def hreflang_check(url: str, page: PageParser, expected: set[str]) -> dict:
|
|
|
134
134
|
return {"ok": not missing and not reciprocal_errors, "missing": missing, "reciprocal_errors": reciprocal_errors, "links": links}
|
|
135
135
|
|
|
136
136
|
|
|
137
|
-
def sitemap_urls(base: str) -> list[str]:
|
|
138
|
-
"""Resolve a sitemap index and return
|
|
137
|
+
def sitemap_urls(base: str) -> tuple[list[str], list[dict[str, str]]]:
|
|
138
|
+
"""Resolve a sitemap index and return page URLs plus raw loc violations."""
|
|
139
139
|
pending = [urljoin(base + "/", "sitemap.xml")]
|
|
140
140
|
pages = []
|
|
141
|
+
violations = []
|
|
141
142
|
seen = set()
|
|
142
143
|
while pending:
|
|
143
144
|
sitemap_url = pending.pop(0)
|
|
@@ -145,12 +146,16 @@ def sitemap_urls(base: str) -> list[str]:
|
|
|
145
146
|
continue
|
|
146
147
|
seen.add(sitemap_url)
|
|
147
148
|
_, _, body = fetch(sitemap_url)
|
|
148
|
-
|
|
149
|
+
raw_locs = [value.strip() for value in re.findall(r"<loc>\s*(.*?)\s*</loc>", body, re.I | re.S)]
|
|
150
|
+
for value in raw_locs:
|
|
151
|
+
if not re.match(r"^https?://[^\s]+$", value, re.I):
|
|
152
|
+
violations.append({"sitemap": sitemap_url, "loc": value, "reason": "sitemap loc must be an absolute HTTP(S) URL"})
|
|
153
|
+
locs = [urljoin(base + "/", value) for value in raw_locs]
|
|
149
154
|
if re.search(r"<sitemapindex\b", body, re.I):
|
|
150
155
|
pending.extend(locs)
|
|
151
156
|
else:
|
|
152
157
|
pages.extend(locs)
|
|
153
|
-
return list(dict.fromkeys(pages))
|
|
158
|
+
return list(dict.fromkeys(pages)), violations
|
|
154
159
|
|
|
155
160
|
|
|
156
161
|
def main() -> int:
|
|
@@ -215,7 +220,8 @@ def main() -> int:
|
|
|
215
220
|
crawl = {"enabled": args.crawl, "passed": True, "pages": []}
|
|
216
221
|
if args.crawl:
|
|
217
222
|
try:
|
|
218
|
-
urls = sitemap_urls(base)
|
|
223
|
+
urls, sitemap_violations = sitemap_urls(base)
|
|
224
|
+
urls = urls[: args.max_pages]
|
|
219
225
|
for page_url in urls:
|
|
220
226
|
try:
|
|
221
227
|
status, content_type, body = fetch(page_url)
|
|
@@ -232,7 +238,8 @@ def main() -> int:
|
|
|
232
238
|
except Exception as exc:
|
|
233
239
|
page_checks = {"url": page_url, "passed": False, "error": type(exc).__name__}
|
|
234
240
|
crawl["pages"].append(page_checks)
|
|
235
|
-
crawl["
|
|
241
|
+
crawl["sitemap_loc_violations"] = sitemap_violations
|
|
242
|
+
crawl["passed"] = not sitemap_violations and bool(urls) and len(crawl["pages"]) == len(urls) and all(item["passed"] for item in crawl["pages"])
|
|
236
243
|
crawl["url_count"] = len(urls)
|
|
237
244
|
except Exception as exc:
|
|
238
245
|
crawl = {"enabled": True, "passed": False, "error": type(exc).__name__, "pages": []}
|