@topy-ai/maggie 0.6.9 → 0.7.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. package/README.md +56 -4
  2. package/bin/maggie.js +11 -7
  3. package/bundled-references/blog-translation-ingestion.md +52 -0
  4. package/bundled-references/maggiedash-dashboard-ui.md +28 -0
  5. package/bundled-skills/maggie-blog/SKILL.md +28 -0
  6. package/bundled-skills/maggie-content-localization/SKILL.md +49 -0
  7. package/bundled-skills/maggie-dash/SKILL.md +52 -0
  8. package/bundled-skills/maggie-deployment/SKILL.md +17 -0
  9. package/bundled-skills/maggie-design/SKILL.md +11 -0
  10. package/bundled-skills/maggie-memory/SKILL.md +7 -0
  11. package/bundled-skills/maggie-ops/SKILL.md +27 -0
  12. package/bundled-skills/maggie-seo-geo/SKILL.md +95 -1
  13. package/bundled-skills/maggie-service-booking/SKILL.md +16 -0
  14. package/bundled-templates/maggiedash/README.md +4 -0
  15. package/bundled-templates/maggiedash/dashboard-ui-contract.json +31 -0
  16. package/bundled-tools/clis/maggie_analytics.py +14 -1
  17. package/bundled-tools/clis/maggie_blog.py +19 -1
  18. package/bundled-tools/clis/maggie_browser_audit.py +78 -0
  19. package/bundled-tools/clis/maggie_dash.py +101 -0
  20. package/bundled-tools/clis/maggie_deployment.py +43 -0
  21. package/bundled-tools/clis/maggie_feedback.py +16 -1
  22. package/bundled-tools/clis/maggie_localization.py +31 -0
  23. package/bundled-tools/clis/maggie_memory.py +4 -1
  24. package/bundled-tools/clis/maggie_ops.py +21 -1
  25. package/bundled-tools/clis/maggie_service_booking.py +6 -3
  26. package/bundled-tools/clis/maggie_sitemap.py +16 -2
  27. package/bundled-tools/clis/site_audit.py +93 -9
  28. package/bundled-tools/runtime/analytics_traffic.py +30 -0
  29. package/bundled-tools/runtime/browser_behavior.py +35 -0
  30. package/bundled-tools/runtime/browser_geometry.js +31 -0
  31. package/bundled-tools/runtime/content_localization.py +63 -1
  32. package/bundled-tools/runtime/dependency_lock.py +43 -0
  33. package/bundled-tools/runtime/integration_state.py +17 -0
  34. package/bundled-tools/runtime/localization_runner.py +113 -0
  35. package/bundled-tools/runtime/maggie_blog.py +66 -1
  36. package/bundled-tools/runtime/maggie_dash_store.py +150 -6
  37. package/bundled-tools/runtime/maggie_dash_ui.py +60 -0
  38. package/bundled-tools/runtime/maggie_memory.py +7 -2
  39. package/bundled-tools/runtime/maggie_sitemap.py +50 -4
  40. package/bundled-tools/runtime/route_imports.py +51 -0
  41. package/bundled-tools/runtime/seed_evidence.py +25 -0
  42. package/bundled-tools/runtime/service_variants.py +152 -0
  43. package/bundled-tools/runtime/site_baseline.py +60 -0
  44. package/package.json +1 -1
  45. package/references/blog-translation-ingestion.md +52 -0
  46. package/references/maggiedash-dashboard-ui.md +28 -0
@@ -8,6 +8,8 @@ import re
8
8
  import shutil
9
9
  from datetime import datetime, timezone
10
10
  from pathlib import Path
11
+ from localization_runner import checkpoint, generate
12
+ from content_localization import valid_locale
11
13
 
12
14
  STATUSES = ("draft", "review", "approved", "published", "archived")
13
15
  DEFAULTS = {
@@ -20,6 +22,8 @@ DEFAULTS = {
20
22
  "pullIntervalMinutes": 120,
21
23
  "maxPerRun": 1,
22
24
  "fallbackImageMode": "gradient",
25
+ "autoTranslateEnabled": False,
26
+ "translationLocales": [],
23
27
  }
24
28
 
25
29
 
@@ -65,7 +69,62 @@ class BlogStore:
65
69
  return json.loads(self.posts_path.read_text(encoding="utf-8"))
66
70
 
67
71
  def _write(self, path: Path, value: object) -> None:
68
- path.write_text(json.dumps(value, indent=2, ensure_ascii=False) + "\n", encoding="utf-8")
72
+ checkpoint(path, value)
73
+
74
+ def reconcile_translations(self) -> list[dict]:
75
+ """Recover missing scheduling intents from persisted source, without re-pull."""
76
+ settings = self.settings()
77
+ if not settings.get("autoTranslateEnabled"):
78
+ return []
79
+ locales = settings.get("translationLocales", [])
80
+ if not isinstance(locales, list) or not locales or any(not isinstance(locale, str) or not valid_locale(locale) for locale in locales):
81
+ raise ValueError("auto translation requires supported translationLocales")
82
+ tasks = []
83
+ for post in self.posts():
84
+ for locale in sorted(set(locales)):
85
+ contract = {"projectId": post["projectId"], "contentId": post["contentId"],
86
+ "sourceRevision": checksum({key: post[key] for key in
87
+ ("title", "excerpt", "body", "topics", "source")}
88
+ | {"images": post.get("images", [])}),
89
+ "targetLocale": locale, "operation": "translate"}
90
+ identity = checksum(contract).split(":")[1]
91
+ path = self.root / "translations" / (identity + ".json")
92
+ if not path.exists():
93
+ self._write(path, {"id": identity, "contract": contract, "status": "pending",
94
+ "attempts": 0, "isIndexable": False})
95
+ tasks.append(json.loads(path.read_text(encoding="utf-8")))
96
+ return tasks
97
+
98
+ def translate_pending(self, adapter, character_budget: int = 8000) -> dict:
99
+ """Single-worker draft translation; every caller uses reconciliation first."""
100
+ tasks = self.reconcile_translations()
101
+ posts = {post["contentId"]: post for post in self.posts()}
102
+ for task in tasks:
103
+ if task["status"] == "succeeded":
104
+ continue
105
+ post = posts[task["contract"]["contentId"]]
106
+ strings = [{"id": key, "text": text} for key, text in (
107
+ ("title", post["title"]), ("excerpt", post["excerpt"]),
108
+ ("body", post["body"].get("value", ""))) if isinstance(text, str) and text.strip()]
109
+ strings.extend({"id": "topic:" + str(index), "text": topic["label"]}
110
+ for index, topic in enumerate(post["topics"]))
111
+ strings.extend({"id": "alt:" + str(index), "text": image["alt"]}
112
+ for index, image in enumerate(post.get("images", []))
113
+ if isinstance(image.get("alt"), str) and image["alt"].strip())
114
+ path = self.root / "translations" / (task["id"] + ".json")
115
+ task.update(status="running", attempts=task["attempts"] + 1)
116
+ self._write(path, task)
117
+ try:
118
+ result = generate(strings, task["contract"],
119
+ self.root / "translations" / "checkpoints" / (task["id"] + ".json"),
120
+ adapter, character_budget)
121
+ task.update(status="succeeded", draft=result["translations"], isIndexable=False)
122
+ task.pop("errorCategory", None)
123
+ except Exception as error:
124
+ task.update(status="failed", errorCategory=type(error).__name__)
125
+ self._write(path, task)
126
+ return {"status": "partial" if any(task["status"] != "succeeded" for task in tasks) else "completed",
127
+ "tasks": [{key: task[key] for key in ("id", "status", "attempts")} for task in tasks]}
69
128
 
70
129
  def _backup(self) -> None:
71
130
  if self.posts_path.exists():
@@ -91,6 +150,8 @@ class BlogStore:
91
150
  "slug": old["slug"] if old else slugify(raw_slug),
92
151
  "title": title,
93
152
  "excerpt": str(item.get("excerpt") or ""),
153
+ "images": [{"src": image.get("src", ""), "alt": image.get("alt", "")}
154
+ for image in item.get("images", []) if isinstance(image, dict)],
94
155
  "body": item.get("body") if isinstance(item.get("body"), dict) else {"format": str(item.get("format") or "markdown"), "value": str(item.get("body") or item.get("content") or "")},
95
156
  "topics": [{"slug": slugify(str(topic)), "label": str(topic)} for topic in item.get("topics", item.get("keywords", []))],
96
157
  # Provider input can suggest a status, but cannot bypass the
@@ -111,8 +172,12 @@ class BlogStore:
111
172
  slugs = [post["slug"] for post in result]
112
173
  if len(slugs) != len(set(slugs)): raise ValueError("slug collision detected; provide unique slugs")
113
174
  self._backup(); self._write(self.posts_path, result)
175
+ translation_tasks = self.reconcile_translations()
114
176
  runs = json.loads(self.runs_path.read_text(encoding="utf-8")) if self.runs_path.exists() else []
115
177
  run = {"id": "pull:" + hashlib.sha256((now() + provider).encode()).hexdigest()[:12], "provider": provider, "status": "completed", "delivered": len(payload), "changed": changed, "finishedAt": now()}
178
+ run["translationTasks"] = len(translation_tasks)
179
+ if any(task["status"] != "succeeded" for task in translation_tasks):
180
+ run["status"] = "partial"
116
181
  runs.append(run); self._write(self.runs_path, runs)
117
182
  return run
118
183
 
@@ -4,10 +4,12 @@
4
4
  from __future__ import annotations
5
5
 
6
6
  import hashlib
7
+ import base64
8
+ import hmac
7
9
  import json
8
10
  import sqlite3
9
11
  import uuid
10
- from datetime import datetime, timezone
12
+ from datetime import datetime, timedelta, timezone
11
13
  from pathlib import Path
12
14
  from typing import Any
13
15
 
@@ -55,6 +57,20 @@ class MaggieDashStore:
55
57
  created_at TEXT NOT NULL, updated_at TEXT NOT NULL,
56
58
  UNIQUE(project_id, slug)
57
59
  );
60
+ CREATE TABLE IF NOT EXISTS maggiedash_document_revisions (
61
+ id TEXT PRIMARY KEY, document_id TEXT NOT NULL REFERENCES maggiedash_documents(id),
62
+ revision INTEGER NOT NULL, title TEXT NOT NULL, slug TEXT NOT NULL,
63
+ locale TEXT NOT NULL, excerpt TEXT NOT NULL, content_json TEXT NOT NULL,
64
+ canonical_url TEXT, provenance_json TEXT NOT NULL, checksum TEXT NOT NULL,
65
+ created_at TEXT NOT NULL, created_by TEXT NOT NULL,
66
+ UNIQUE(document_id, revision)
67
+ );
68
+ CREATE TABLE IF NOT EXISTS maggiedash_redirects (
69
+ id TEXT PRIMARY KEY, project_id TEXT NOT NULL REFERENCES maggiedash_projects(id),
70
+ from_path TEXT NOT NULL, to_path TEXT NOT NULL, status_code INTEGER NOT NULL DEFAULT 301,
71
+ reason TEXT NOT NULL, actor_id TEXT NOT NULL, created_at TEXT NOT NULL,
72
+ UNIQUE(project_id, from_path)
73
+ );
58
74
  CREATE TABLE IF NOT EXISTS maggiedash_approvals (
59
75
  id TEXT PRIMARY KEY, document_id TEXT NOT NULL REFERENCES maggiedash_documents(id),
60
76
  from_status TEXT NOT NULL, to_status TEXT NOT NULL, actor_id TEXT NOT NULL,
@@ -68,6 +84,15 @@ class MaggieDashStore:
68
84
  );
69
85
  """
70
86
  )
87
+ self._ensure_column("maggiedash_documents", "trashed_at", "TEXT")
88
+ self._ensure_column("maggiedash_documents", "trashed_from_status", "TEXT")
89
+ self._ensure_column("maggiedash_documents", "publish_at", "TEXT")
90
+ self.connection.commit()
91
+
92
+ def _ensure_column(self, table: str, column: str, definition: str) -> None:
93
+ columns = {row[1] for row in self.connection.execute(f"PRAGMA table_info({table})")}
94
+ if column not in columns:
95
+ self.connection.execute(f"ALTER TABLE {table} ADD COLUMN {column} {definition}")
71
96
 
72
97
  def close(self) -> None:
73
98
  self.connection.close()
@@ -97,22 +122,27 @@ class MaggieDashStore:
97
122
  raise ValueError(f"invalid document status: {status}")
98
123
  stable = {key: value for key, value in document.items() if key not in {"provenance", "checksum"}}
99
124
  document_checksum = document.get("provenance", {}).get("checksum") or checksum(stable)
100
- existing = self.connection.execute("SELECT status FROM maggiedash_documents WHERE id=?", (document["id"],)).fetchone()
125
+ existing = self.connection.execute("SELECT * FROM maggiedash_documents WHERE id=?", (document["id"],)).fetchone()
101
126
  if existing and existing["status"] != status:
102
127
  raise ValueError("status changes must use transition()")
128
+ if existing and existing["checksum"] != document_checksum:
129
+ self._store_revision(existing, actor_id)
103
130
  self.connection.execute(
104
131
  """INSERT INTO maggiedash_documents
105
- (id,project_id,kind,title,slug,locale,status,excerpt,content_json,canonical_url,provenance_json,checksum,created_at,updated_at)
106
- VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?)
132
+ (id,project_id,kind,title,slug,locale,status,excerpt,content_json,canonical_url,provenance_json,checksum,created_at,updated_at,publish_at)
133
+ VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)
107
134
  ON CONFLICT(id) DO UPDATE SET title=excluded.title, slug=excluded.slug,
108
135
  locale=excluded.locale, excerpt=excluded.excerpt, content_json=excluded.content_json,
109
136
  canonical_url=excluded.canonical_url, provenance_json=excluded.provenance_json,
110
- checksum=excluded.checksum, updated_at=excluded.updated_at""",
137
+ checksum=excluded.checksum, updated_at=excluded.updated_at,
138
+ publish_at=excluded.publish_at, trashed_at=NULL, trashed_from_status=NULL""",
111
139
  (document["id"], project_id, document["kind"], document["title"], document["slug"], document["locale"],
112
140
  status, document.get("excerpt", ""), json.dumps(document["content"], ensure_ascii=False),
113
141
  document.get("canonicalUrl"), json.dumps(document["provenance"], ensure_ascii=False), document_checksum,
114
- timestamp, timestamp),
142
+ timestamp, timestamp, document.get("publishAt")),
115
143
  )
144
+ if not existing:
145
+ self._store_revision(self.connection.execute("SELECT * FROM maggiedash_documents WHERE id=?", (document["id"],)).fetchone(), actor_id)
116
146
  self._audit(project_id, "document.upserted", "document", document["id"], actor_id, "draft content stored")
117
147
  self.connection.commit()
118
148
  return self.get_document(project_id, document["id"]) or {}
@@ -125,8 +155,122 @@ class MaggieDashStore:
125
155
  result["content"] = json.loads(result.pop("content_json"))
126
156
  result["provenance"] = json.loads(result.pop("provenance_json"))
127
157
  result["canonicalUrl"] = result.pop("canonical_url")
158
+ result["trashedAt"] = result.pop("trashed_at")
159
+ result["trashedFromStatus"] = result.pop("trashed_from_status")
160
+ result["publishAt"] = result.pop("publish_at")
128
161
  return result
129
162
 
163
+ def _store_revision(self, row: sqlite3.Row, actor_id: str) -> None:
164
+ revision = self.connection.execute(
165
+ "SELECT COALESCE(MAX(revision), 0) + 1 FROM maggiedash_document_revisions WHERE document_id=?",
166
+ (row["id"],),
167
+ ).fetchone()[0]
168
+ self.connection.execute(
169
+ "INSERT INTO maggiedash_document_revisions VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?)",
170
+ (str(uuid.uuid4()), row["id"], revision, row["title"], row["slug"], row["locale"], row["excerpt"],
171
+ row["content_json"], row["canonical_url"], row["provenance_json"], row["checksum"], row["updated_at"], actor_id),
172
+ )
173
+
174
+ def list_revisions(self, project_id: str, document_id: str) -> list[dict[str, Any]]:
175
+ if not self.get_document(project_id, document_id):
176
+ raise ValueError("document not found")
177
+ rows = self.connection.execute(
178
+ "SELECT * FROM maggiedash_document_revisions WHERE document_id=? ORDER BY revision DESC", (document_id,)
179
+ ).fetchall()
180
+ return [dict(row) for row in rows]
181
+
182
+ def trash(self, project_id: str, document_id: str, actor_id: str, reason: str) -> dict[str, Any]:
183
+ row = self.connection.execute("SELECT status, trashed_at FROM maggiedash_documents WHERE project_id=? AND id=?", (project_id, document_id)).fetchone()
184
+ if not row:
185
+ raise ValueError("document not found")
186
+ if row["trashed_at"]:
187
+ return self.get_document(project_id, document_id) or {}
188
+ timestamp = now()
189
+ self.connection.execute("UPDATE maggiedash_documents SET trashed_at=?, trashed_from_status=?, updated_at=? WHERE id=?", (timestamp, row["status"], timestamp, document_id))
190
+ self._audit(project_id, "document.trashed", "document", document_id, actor_id, reason)
191
+ self.connection.commit()
192
+ return self.get_document(project_id, document_id) or {}
193
+
194
+ def restore(self, project_id: str, document_id: str, actor_id: str, reason: str) -> dict[str, Any]:
195
+ row = self.connection.execute("SELECT trashed_at, trashed_from_status FROM maggiedash_documents WHERE project_id=? AND id=?", (project_id, document_id)).fetchone()
196
+ if not row:
197
+ raise ValueError("document not found")
198
+ if not row["trashed_at"]:
199
+ return self.get_document(project_id, document_id) or {}
200
+ status = row["trashed_from_status"] if row["trashed_from_status"] in STATUSES else "draft"
201
+ timestamp = now()
202
+ self.connection.execute("UPDATE maggiedash_documents SET status=?, trashed_at=NULL, trashed_from_status=NULL, updated_at=? WHERE id=?", (status, timestamp, document_id))
203
+ self._audit(project_id, "document.restored", "document", document_id, actor_id, reason)
204
+ self.connection.commit()
205
+ return self.get_document(project_id, document_id) or {}
206
+
207
+ def schedule_publish(self, project_id: str, document_id: str, publish_at: str, actor_id: str, reason: str) -> dict[str, Any]:
208
+ try:
209
+ target = datetime.fromisoformat(publish_at.replace("Z", "+00:00"))
210
+ except ValueError as exc:
211
+ raise ValueError("publish_at must be an ISO-8601 timestamp") from exc
212
+ if target.tzinfo is None:
213
+ raise ValueError("publish_at must include a timezone")
214
+ document = self.get_document(project_id, document_id)
215
+ if not document:
216
+ raise ValueError("document not found")
217
+ if document["status"] not in {"approved", "scheduled"}:
218
+ raise ValueError("only approved documents can be scheduled")
219
+ if document["status"] == "approved":
220
+ self.transition(project_id, document_id, "scheduled", actor_id, reason)
221
+ self.connection.execute("UPDATE maggiedash_documents SET publish_at=?, updated_at=? WHERE id=?", (publish_at, now(), document_id))
222
+ self._audit(project_id, "document.publish_scheduled", "document", document_id, actor_id, reason)
223
+ self.connection.commit()
224
+ return self.get_document(project_id, document_id) or {}
225
+
226
+ def duplicate(self, project_id: str, document_id: str, new_id: str, new_slug: str, actor_id: str, reason: str) -> dict[str, Any]:
227
+ source = self.get_document(project_id, document_id)
228
+ if not source:
229
+ raise ValueError("document not found")
230
+ if self.connection.execute("SELECT 1 FROM maggiedash_documents WHERE project_id=? AND slug=?", (project_id, new_slug)).fetchone():
231
+ raise ValueError("new slug already exists")
232
+ copy = dict(source)
233
+ copy.update({"id": new_id, "slug": new_slug, "status": "draft", "provenance": {**source["provenance"], "duplicatedFrom": document_id}})
234
+ copy.pop("trashedAt", None); copy.pop("trashedFromStatus", None); copy.pop("publishAt", None)
235
+ result = self.put_document(project_id, copy, actor_id)
236
+ self._audit(project_id, "document.duplicated", "document", new_id, actor_id, reason)
237
+ self.connection.commit()
238
+ return result
239
+
240
+ def add_redirect(self, project_id: str, from_path: str, to_path: str, actor_id: str, reason: str, status_code: int = 301) -> dict[str, Any]:
241
+ if not from_path.startswith("/") or not to_path.startswith("/") or from_path == to_path:
242
+ raise ValueError("redirect paths must be distinct absolute paths")
243
+ if status_code not in {301, 302, 307, 308}:
244
+ raise ValueError("status_code must be 301, 302, 307 or 308")
245
+ record = (str(uuid.uuid4()), project_id, from_path, to_path, status_code, reason, actor_id, now())
246
+ self.connection.execute("INSERT INTO maggiedash_redirects VALUES (?,?,?,?,?,?,?,?) ON CONFLICT(project_id,from_path) DO UPDATE SET to_path=excluded.to_path, status_code=excluded.status_code, reason=excluded.reason, actor_id=excluded.actor_id, created_at=excluded.created_at", record)
247
+ self._audit(project_id, "redirect.upserted", "redirect", from_path, actor_id, reason)
248
+ self.connection.commit()
249
+ row = self.connection.execute("SELECT * FROM maggiedash_redirects WHERE project_id=? AND from_path=?", (project_id, from_path)).fetchone()
250
+ return dict(row)
251
+
252
+ def issue_preview(self, project_id: str, document_id: str, secret: str, ttl_seconds: int = 900) -> dict[str, Any]:
253
+ if not secret:
254
+ raise ValueError("preview secret is required")
255
+ if not self.get_document(project_id, document_id):
256
+ raise ValueError("document not found")
257
+ expires = int((datetime.now(timezone.utc) + timedelta(seconds=ttl_seconds)).timestamp())
258
+ payload = f"{project_id}:{document_id}:{expires}".encode()
259
+ encoded = base64.urlsafe_b64encode(payload).decode().rstrip("=")
260
+ signature = hmac.new(secret.encode(), encoded.encode(), hashlib.sha256).hexdigest()
261
+ return {"token": f"{encoded}.{signature}", "expiresAt": datetime.fromtimestamp(expires, timezone.utc).isoformat().replace("+00:00", "Z")}
262
+
263
+ @staticmethod
264
+ def verify_preview(token: str, secret: str, project_id: str, document_id: str) -> bool:
265
+ try:
266
+ encoded, signature = token.split(".", 1)
267
+ expected = hmac.new(secret.encode(), encoded.encode(), hashlib.sha256).hexdigest()
268
+ payload = base64.urlsafe_b64decode(encoded + "=" * (-len(encoded) % 4)).decode()
269
+ token_project, token_document, expires = payload.split(":", 2)
270
+ return hmac.compare_digest(signature, expected) and token_project == project_id and token_document == document_id and int(expires) >= int(datetime.now(timezone.utc).timestamp())
271
+ except (ValueError, TypeError, UnicodeDecodeError):
272
+ return False
273
+
130
274
  def transition(self, project_id: str, document_id: str, target: str, actor_id: str, reason: str) -> dict[str, Any]:
131
275
  if target not in STATUSES:
132
276
  raise ValueError(f"invalid target status: {target}")
@@ -0,0 +1,60 @@
1
+ """Validate the provider-neutral MaggieDash dashboard chrome contract."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ from pathlib import Path
7
+ from typing import Iterable
8
+
9
+
10
+ SCHEMA = "maggiedash-dashboard-ui.v1"
11
+
12
+
13
+ def _component_source(name: str, source: str) -> str:
14
+ match = re.search(rf"(?:function\s+{re.escape(name)}\s*\([^)]*\)|const\s+{re.escape(name)}\s*=\s*\([^)]*\))(?P<body>[\s\S]{{0,12000}})", source)
15
+ return match.group(0) if match else ""
16
+
17
+
18
+ def validate(contract: object, sources: Iterable[str] = ()) -> dict:
19
+ errors: list[str] = []
20
+ if not isinstance(contract, dict):
21
+ return {"schemaVersion": SCHEMA, "passed": False, "errors": ["contract must be an object"]}
22
+ if contract.get("schemaVersion") != SCHEMA:
23
+ errors.append(f"schemaVersion must be {SCHEMA}")
24
+ shell = contract.get("shell")
25
+ if not isinstance(shell, dict):
26
+ errors.append("shell is required")
27
+ elif shell.get("reference") != "users-workspace" or shell.get("navigation") != "sidebar-plus-content-tabs":
28
+ errors.append("shell must declare the users-workspace reference and sidebar-plus-content-tabs navigation")
29
+ components = contract.get("components")
30
+ if not isinstance(components, list) or not components:
31
+ errors.append("components must be a non-empty list")
32
+ components = []
33
+ source = "\n".join(str(item) for item in sources)
34
+ for component in components:
35
+ if not isinstance(component, dict) or not component.get("name"):
36
+ errors.append("each component needs a name")
37
+ continue
38
+ name = str(component["name"])
39
+ component_source = _component_source(name, source)
40
+ if source and not component_source:
41
+ errors.append(f"component source is missing: {name}")
42
+ continue
43
+ props = [str(prop) for prop in component.get("props", [])]
44
+ forbidden = [str(prop) for prop in component.get("forbiddenProps", [])]
45
+ if component_source:
46
+ signature = component_source.split("=>", 1)[0] if "=>" in component_source else component_source[:500]
47
+ for prop in props:
48
+ if len(re.findall(rf"\b{re.escape(prop)}\b", component_source)) < 2:
49
+ errors.append(f"{name}.{prop} is declared but not evidenced in rendered output")
50
+ for prop in forbidden:
51
+ if re.search(rf"\b{re.escape(prop)}\b", signature):
52
+ errors.append(f"{name} declares forbidden dead prop: {prop}")
53
+ return {"schemaVersion": SCHEMA, "passed": not errors, "errors": errors, "components": [item.get("name") for item in components if isinstance(item, dict)]}
54
+
55
+
56
+ def load_and_validate(contract_path: Path, source_paths: Iterable[Path] = ()) -> dict:
57
+ import json
58
+ contract = json.loads(contract_path.read_text(encoding="utf-8"))
59
+ sources = [path.read_text(encoding="utf-8", errors="replace") for path in source_paths]
60
+ return validate(contract, sources)
@@ -171,12 +171,17 @@ def _active(item: dict[str, Any], project_id: str = "") -> bool:
171
171
  return not expires or expires > now()
172
172
 
173
173
 
174
- def relevant(project: str | Path, *, skill: str = "", query: str = "", project_id: str = "") -> dict[str, list[dict[str, Any]]]:
174
+ def relevant(project: str | Path, *, skill: str = "", query: str = "", project_id: str = "", status: str | None = None) -> dict[str, list[dict[str, Any]]]:
175
175
  terms = set(re.findall(r"[a-z0-9_-]+", f"{skill} {query}".lower()))
176
176
  result: dict[str, list[dict[str, Any]]] = {kind: [] for kind in KINDS}
177
177
  for kind in KINDS:
178
178
  for item in read(project, kind)["items"]:
179
- if kind != "errors" and not _active(item, project_id or root(project).name):
179
+ if status is not None:
180
+ if item.get("status") != status:
181
+ continue
182
+ if item.get("scope") == "project" and item.get("projectId") not in {project_id or root(project).name, ""}:
183
+ continue
184
+ elif kind != "errors" and not _active(item, project_id or root(project).name):
180
185
  continue
181
186
  haystack = " ".join(str(item.get(key, "")) for key in ("type", "title", "rule", "problem", "solution", "prevention", "skill", "message", "fingerprint", "triggers")).lower()
182
187
  if not terms or terms & set(re.findall(r"[a-z0-9_-]+", haystack)):
@@ -55,6 +55,12 @@ def filter_routes(routes: list[dict[str, str]], origin: str, content_types: set[
55
55
  reason = "content type not enabled"
56
56
  elif not absolute_url(url, origin):
57
57
  reason = "URL must be an absolute HTTP(S) URL on the approved origin"
58
+ elif route.get("indexable") is False:
59
+ reason = "route is explicitly non-indexable"
60
+ elif route.get("canonicalUrl") and route.get("canonicalUrl") != url:
61
+ reason = "route canonical URL does not self-canonicalize"
62
+ elif route.get("searchable") is False:
63
+ reason = "route is explicitly not searchable"
58
64
  elif EXCLUDED_PATHS.search(parsed.path) or any(key in parse_qs(parsed.query) for key in ("q", "search", "filter", "page")):
59
65
  reason = "search, filter, or non-canonical route"
60
66
  elif route.get("type", "").lower() in {"rss", "atom", "feed"}:
@@ -86,12 +92,12 @@ def build_plan(routes: list[dict[str, str]], origin: str, content_types: set[str
86
92
  for content_type in sorted(content_types):
87
93
  chunks = []
88
94
  entries = groups.get(content_type, []) or []
89
- for index in range(0, max(len(entries), 1), chunk_target):
95
+ for index in range(0, len(entries), chunk_target):
90
96
  chunk_entries = entries[index:index + chunk_target]
91
97
  filename = f"sitemap-{content_type}-{index // chunk_target + 1}.xml"
92
98
  xml = xml_file(chunk_entries)
93
99
  url = origin.rstrip("/") + "/" + filename
94
- chunks.append({"filename": filename, "url": url, "entries": len(chunk_entries), "bytes": len(xml.encode()), "sha256": hashlib.sha256(xml.encode()).hexdigest(), "xml": xml})
100
+ chunks.append({"filename": filename, "url": url, "entries": len(chunk_entries), "routes": chunk_entries, "bytes": len(xml.encode()), "sha256": hashlib.sha256(xml.encode()).hexdigest(), "xml": xml})
95
101
  sitemap_urls.append(url)
96
102
  group_plans.append({"contentType": content_type, "chunks": chunks})
97
103
  index_xml = '<?xml version="1.0" encoding="UTF-8"?>\n<sitemapindex xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">' + "".join(f"<sitemap><loc>{escape(url)}</loc></sitemap>" for url in sitemap_urls) + "</sitemapindex>\n"
@@ -103,7 +109,7 @@ def build_plan(routes: list[dict[str, str]], origin: str, content_types: set[str
103
109
  return plan
104
110
 
105
111
 
106
- def validate_plan_data(plan: dict) -> dict:
112
+ def validate_plan_data(plan: dict, strict_semantic: bool = False) -> dict:
107
113
  errors = []
108
114
  warnings = []
109
115
  origin = plan.get("origin", "")
@@ -117,6 +123,21 @@ def validate_plan_data(plan: dict) -> dict:
117
123
  errors.append(f"chunk exceeds URL limit: {chunk.get('filename')}")
118
124
  if chunk.get("bytes", 0) > HARD_BYTES_LIMIT:
119
125
  errors.append(f"chunk exceeds byte limit: {chunk.get('filename')}")
126
+ for route in chunk.get("routes", []):
127
+ if route.get("indexable") is False or route.get("searchable") is False:
128
+ errors.append(f"non-indexable/searchable route was emitted: {route.get('url')}")
129
+ if route.get("canonicalUrl") and route.get("canonicalUrl") != route.get("url"):
130
+ errors.append(f"non-self-canonical route was emitted: {route.get('url')}")
131
+ if "contentWords" in route and int(route.get("contentWords") or 0) < 1:
132
+ errors.append(f"route has no searchable content evidence: {route.get('url')}")
133
+ if "title" in route and not str(route.get("title") or "").strip():
134
+ errors.append(f"route has no title evidence: {route.get('url')}")
135
+ if route.get("lastmod") and route.get("lastmodSource") in {"updated_at", "sync", "pull", "deploy"}:
136
+ errors.append(f"lastmod is derived from an operational timestamp: {route.get('url')}")
137
+ if route.get("lastmod") and route.get("lastmodSource") not in {None, "content-change", "source-revision"}:
138
+ errors.append(f"lastmod provenance is not a content change: {route.get('url')}")
139
+ if route.get("lastmod") and strict_semantic and route.get("lastmodSource") not in {"content-change", "source-revision"}:
140
+ errors.append(f"lastmod evidence is required for strict semantic validation: {route.get('url')}")
120
141
  try:
121
142
  root = ElementTree.fromstring(chunk.get("xml", ""))
122
143
  locs = [element.text or "" for element in root.iter() if element.tag.endswith("loc")]
@@ -140,5 +161,30 @@ def validate_plan_data(plan: dict) -> dict:
140
161
  except ElementTree.ParseError:
141
162
  errors.append("sitemap index is not valid XML")
142
163
  if not expected_urls:
143
- warnings.append("no sitemap chunks generated")
164
+ warnings.append("no sitemap chunks generated; empty content types are not advertised")
165
+ if strict_semantic:
166
+ for group in plan.get("groups", []):
167
+ for chunk in group.get("chunks", []):
168
+ for route in chunk.get("routes", []):
169
+ if not str(route.get("title") or "").strip(): warnings.append(f"semantic title evidence missing: {route.get('url')}")
170
+ if "contentWords" not in route: warnings.append(f"semantic content evidence missing: {route.get('url')}")
171
+ if warnings:
172
+ errors.extend(warnings)
144
173
  return {"status": "pass" if not errors else "fail", "errors": errors, "warnings": warnings}
174
+
175
+
176
+ def agent_files(routes: list[dict[str, str]], origin: str, locale: str = "en") -> dict[str, str]:
177
+ """Generate compact, locale-aware agent discovery documents."""
178
+ visible = [route for route in routes if route.get("indexable") is not False and route.get("searchable") is not False and absolute_url(route.get("url", ""), origin)]
179
+ visible.sort(key=lambda route: route.get("url", ""))
180
+ lines = [f"# {origin} ({locale})", "", "This file lists indexable public routes for agent discovery.", "", "## Routes"]
181
+ for route in visible:
182
+ label = route.get("title") or route.get("url", "").rstrip("/").rsplit("/", 1)[-1] or "home"
183
+ lines.append(f"- [{label}]({route['url']})")
184
+ sitemap = [f"# Sitemap ({locale})", "", f"- XML sitemap: {origin.rstrip('/')}/sitemap.xml", "", "## Public routes"]
185
+ sitemap.extend(f"- {route['url']}" for route in visible)
186
+ insights = [f"# Content insights ({locale})", "", f"- Indexable routes: {len(visible)}", f"- Content types: {', '.join(sorted({str(route.get('type', 'unknown')) for route in visible})) or 'none'}", "", "## Freshness"]
187
+ for route in visible:
188
+ if route.get("lastmod"):
189
+ insights.append(f"- {route['url']}: last content change {route['lastmod']}")
190
+ return {"llms.txt": "\n".join(lines) + "\n", "sitemap.md": "\n".join(sitemap) + "\n", "insights.md": "\n".join(insights) + "\n"}
@@ -0,0 +1,51 @@
1
+ """Resolve route components from source imports, never from sibling basenames."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import hashlib
6
+ import re
7
+ from pathlib import Path
8
+
9
+
10
+ IMPORT_RE = re.compile(r"(?:import\s+(?:[^;\n]*?\s+from\s+)?|export\s+[^;\n]*?\s+from\s+)[\"']([^\"']+)[\"']")
11
+ EXTENSIONS = ("", ".astro", ".tsx", ".jsx", ".ts", ".js", ".vue", ".svelte", ".html")
12
+
13
+
14
+ def resolve_import(route: Path, specifier: str) -> Path | None:
15
+ if not specifier.startswith("."):
16
+ return None
17
+ base = (route.parent / specifier).resolve()
18
+ candidates = [base]
19
+ candidates.extend(Path(str(base) + extension) for extension in EXTENSIONS[1:])
20
+ candidates.extend(base / f"index{extension}" for extension in EXTENSIONS)
21
+ for candidate in candidates:
22
+ if candidate.is_file():
23
+ return candidate
24
+ return None
25
+
26
+
27
+ def imported_components(route: Path) -> list[dict[str, str | None]]:
28
+ """Return only source imports that resolve inside the project.
29
+
30
+ Bare package imports are intentionally recorded without a local path; they
31
+ are dependencies, not route-local components. The route source remains
32
+ the authority, so a same-named sibling file is never selected implicitly.
33
+ """
34
+ try:
35
+ source = route.read_text(encoding="utf-8", errors="replace")
36
+ except OSError:
37
+ return []
38
+ result = []
39
+ seen = set()
40
+ for specifier in IMPORT_RE.findall(source):
41
+ resolved = resolve_import(route, specifier)
42
+ key = (specifier, str(resolved) if resolved else None)
43
+ if key in seen:
44
+ continue
45
+ seen.add(key)
46
+ result.append({
47
+ "specifier": specifier,
48
+ "path": str(resolved) if resolved else None,
49
+ "fingerprint": hashlib.sha256(resolved.read_bytes()).hexdigest() if resolved else None,
50
+ })
51
+ return result
@@ -0,0 +1,25 @@
1
+ """Validation for explicit, sanitized development fixture evidence."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+
7
+
8
+ SECRET = re.compile(r"(?:secret|token|password|api[_-]?key|database_url)", re.I)
9
+
10
+
11
+ def validate_manifest(value: object) -> dict:
12
+ errors = []
13
+ if not isinstance(value, dict):
14
+ return {"schemaVersion": "maggie-seed-manifest.v1", "passed": False, "errors": ["manifest must be an object"]}
15
+ if value.get("schemaVersion") != "maggie-seed-manifest.v1": errors.append("unsupported schemaVersion")
16
+ if value.get("optIn") is not True: errors.append("optIn must be true")
17
+ if value.get("sanitized") is not True: errors.append("sanitized must be true")
18
+ fixtures = value.get("fixtures")
19
+ if not isinstance(fixtures, list) or not fixtures: errors.append("fixtures must be a non-empty list")
20
+ for item in fixtures if isinstance(fixtures, list) else []:
21
+ if not isinstance(item, dict) or not item.get("name") or not item.get("rows"):
22
+ errors.append("each fixture requires name and non-empty rows")
23
+ if SECRET.search(str(item)):
24
+ errors.append("fixture contains a secret-like field")
25
+ return {"schemaVersion": "maggie-seed-manifest.v1", "passed": not errors, "errors": errors, "fixtureCount": len(fixtures) if isinstance(fixtures, list) else 0}