@topy-ai/maggie 0.6.9 → 0.7.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +56 -4
- package/bin/maggie.js +11 -7
- package/bundled-references/blog-translation-ingestion.md +52 -0
- package/bundled-references/maggiedash-dashboard-ui.md +28 -0
- package/bundled-skills/maggie-blog/SKILL.md +28 -0
- package/bundled-skills/maggie-content-localization/SKILL.md +49 -0
- package/bundled-skills/maggie-dash/SKILL.md +52 -0
- package/bundled-skills/maggie-deployment/SKILL.md +17 -0
- package/bundled-skills/maggie-design/SKILL.md +11 -0
- package/bundled-skills/maggie-memory/SKILL.md +7 -0
- package/bundled-skills/maggie-ops/SKILL.md +27 -0
- package/bundled-skills/maggie-seo-geo/SKILL.md +95 -1
- package/bundled-skills/maggie-service-booking/SKILL.md +16 -0
- package/bundled-templates/maggiedash/README.md +4 -0
- package/bundled-templates/maggiedash/dashboard-ui-contract.json +31 -0
- package/bundled-tools/clis/maggie_analytics.py +14 -1
- package/bundled-tools/clis/maggie_blog.py +19 -1
- package/bundled-tools/clis/maggie_browser_audit.py +78 -0
- package/bundled-tools/clis/maggie_dash.py +101 -0
- package/bundled-tools/clis/maggie_deployment.py +43 -0
- package/bundled-tools/clis/maggie_feedback.py +16 -1
- package/bundled-tools/clis/maggie_localization.py +31 -0
- package/bundled-tools/clis/maggie_memory.py +4 -1
- package/bundled-tools/clis/maggie_ops.py +21 -1
- package/bundled-tools/clis/maggie_service_booking.py +6 -3
- package/bundled-tools/clis/maggie_sitemap.py +16 -2
- package/bundled-tools/clis/site_audit.py +93 -9
- package/bundled-tools/runtime/analytics_traffic.py +30 -0
- package/bundled-tools/runtime/browser_behavior.py +35 -0
- package/bundled-tools/runtime/browser_geometry.js +31 -0
- package/bundled-tools/runtime/content_localization.py +63 -1
- package/bundled-tools/runtime/dependency_lock.py +43 -0
- package/bundled-tools/runtime/integration_state.py +17 -0
- package/bundled-tools/runtime/localization_runner.py +113 -0
- package/bundled-tools/runtime/maggie_blog.py +66 -1
- package/bundled-tools/runtime/maggie_dash_store.py +150 -6
- package/bundled-tools/runtime/maggie_dash_ui.py +60 -0
- package/bundled-tools/runtime/maggie_memory.py +7 -2
- package/bundled-tools/runtime/maggie_sitemap.py +50 -4
- package/bundled-tools/runtime/route_imports.py +51 -0
- package/bundled-tools/runtime/seed_evidence.py +25 -0
- package/bundled-tools/runtime/service_variants.py +152 -0
- package/bundled-tools/runtime/site_baseline.py +60 -0
- package/package.json +1 -1
- package/references/blog-translation-ingestion.md +52 -0
- package/references/maggiedash-dashboard-ui.md +28 -0
|
@@ -8,6 +8,8 @@ import re
|
|
|
8
8
|
import shutil
|
|
9
9
|
from datetime import datetime, timezone
|
|
10
10
|
from pathlib import Path
|
|
11
|
+
from localization_runner import checkpoint, generate
|
|
12
|
+
from content_localization import valid_locale
|
|
11
13
|
|
|
12
14
|
STATUSES = ("draft", "review", "approved", "published", "archived")
|
|
13
15
|
DEFAULTS = {
|
|
@@ -20,6 +22,8 @@ DEFAULTS = {
|
|
|
20
22
|
"pullIntervalMinutes": 120,
|
|
21
23
|
"maxPerRun": 1,
|
|
22
24
|
"fallbackImageMode": "gradient",
|
|
25
|
+
"autoTranslateEnabled": False,
|
|
26
|
+
"translationLocales": [],
|
|
23
27
|
}
|
|
24
28
|
|
|
25
29
|
|
|
@@ -65,7 +69,62 @@ class BlogStore:
|
|
|
65
69
|
return json.loads(self.posts_path.read_text(encoding="utf-8"))
|
|
66
70
|
|
|
67
71
|
def _write(self, path: Path, value: object) -> None:
|
|
68
|
-
path
|
|
72
|
+
checkpoint(path, value)
|
|
73
|
+
|
|
74
|
+
def reconcile_translations(self) -> list[dict]:
|
|
75
|
+
"""Recover missing scheduling intents from persisted source, without re-pull."""
|
|
76
|
+
settings = self.settings()
|
|
77
|
+
if not settings.get("autoTranslateEnabled"):
|
|
78
|
+
return []
|
|
79
|
+
locales = settings.get("translationLocales", [])
|
|
80
|
+
if not isinstance(locales, list) or not locales or any(not isinstance(locale, str) or not valid_locale(locale) for locale in locales):
|
|
81
|
+
raise ValueError("auto translation requires supported translationLocales")
|
|
82
|
+
tasks = []
|
|
83
|
+
for post in self.posts():
|
|
84
|
+
for locale in sorted(set(locales)):
|
|
85
|
+
contract = {"projectId": post["projectId"], "contentId": post["contentId"],
|
|
86
|
+
"sourceRevision": checksum({key: post[key] for key in
|
|
87
|
+
("title", "excerpt", "body", "topics", "source")}
|
|
88
|
+
| {"images": post.get("images", [])}),
|
|
89
|
+
"targetLocale": locale, "operation": "translate"}
|
|
90
|
+
identity = checksum(contract).split(":")[1]
|
|
91
|
+
path = self.root / "translations" / (identity + ".json")
|
|
92
|
+
if not path.exists():
|
|
93
|
+
self._write(path, {"id": identity, "contract": contract, "status": "pending",
|
|
94
|
+
"attempts": 0, "isIndexable": False})
|
|
95
|
+
tasks.append(json.loads(path.read_text(encoding="utf-8")))
|
|
96
|
+
return tasks
|
|
97
|
+
|
|
98
|
+
def translate_pending(self, adapter, character_budget: int = 8000) -> dict:
|
|
99
|
+
"""Single-worker draft translation; every caller uses reconciliation first."""
|
|
100
|
+
tasks = self.reconcile_translations()
|
|
101
|
+
posts = {post["contentId"]: post for post in self.posts()}
|
|
102
|
+
for task in tasks:
|
|
103
|
+
if task["status"] == "succeeded":
|
|
104
|
+
continue
|
|
105
|
+
post = posts[task["contract"]["contentId"]]
|
|
106
|
+
strings = [{"id": key, "text": text} for key, text in (
|
|
107
|
+
("title", post["title"]), ("excerpt", post["excerpt"]),
|
|
108
|
+
("body", post["body"].get("value", ""))) if isinstance(text, str) and text.strip()]
|
|
109
|
+
strings.extend({"id": "topic:" + str(index), "text": topic["label"]}
|
|
110
|
+
for index, topic in enumerate(post["topics"]))
|
|
111
|
+
strings.extend({"id": "alt:" + str(index), "text": image["alt"]}
|
|
112
|
+
for index, image in enumerate(post.get("images", []))
|
|
113
|
+
if isinstance(image.get("alt"), str) and image["alt"].strip())
|
|
114
|
+
path = self.root / "translations" / (task["id"] + ".json")
|
|
115
|
+
task.update(status="running", attempts=task["attempts"] + 1)
|
|
116
|
+
self._write(path, task)
|
|
117
|
+
try:
|
|
118
|
+
result = generate(strings, task["contract"],
|
|
119
|
+
self.root / "translations" / "checkpoints" / (task["id"] + ".json"),
|
|
120
|
+
adapter, character_budget)
|
|
121
|
+
task.update(status="succeeded", draft=result["translations"], isIndexable=False)
|
|
122
|
+
task.pop("errorCategory", None)
|
|
123
|
+
except Exception as error:
|
|
124
|
+
task.update(status="failed", errorCategory=type(error).__name__)
|
|
125
|
+
self._write(path, task)
|
|
126
|
+
return {"status": "partial" if any(task["status"] != "succeeded" for task in tasks) else "completed",
|
|
127
|
+
"tasks": [{key: task[key] for key in ("id", "status", "attempts")} for task in tasks]}
|
|
69
128
|
|
|
70
129
|
def _backup(self) -> None:
|
|
71
130
|
if self.posts_path.exists():
|
|
@@ -91,6 +150,8 @@ class BlogStore:
|
|
|
91
150
|
"slug": old["slug"] if old else slugify(raw_slug),
|
|
92
151
|
"title": title,
|
|
93
152
|
"excerpt": str(item.get("excerpt") or ""),
|
|
153
|
+
"images": [{"src": image.get("src", ""), "alt": image.get("alt", "")}
|
|
154
|
+
for image in item.get("images", []) if isinstance(image, dict)],
|
|
94
155
|
"body": item.get("body") if isinstance(item.get("body"), dict) else {"format": str(item.get("format") or "markdown"), "value": str(item.get("body") or item.get("content") or "")},
|
|
95
156
|
"topics": [{"slug": slugify(str(topic)), "label": str(topic)} for topic in item.get("topics", item.get("keywords", []))],
|
|
96
157
|
# Provider input can suggest a status, but cannot bypass the
|
|
@@ -111,8 +172,12 @@ class BlogStore:
|
|
|
111
172
|
slugs = [post["slug"] for post in result]
|
|
112
173
|
if len(slugs) != len(set(slugs)): raise ValueError("slug collision detected; provide unique slugs")
|
|
113
174
|
self._backup(); self._write(self.posts_path, result)
|
|
175
|
+
translation_tasks = self.reconcile_translations()
|
|
114
176
|
runs = json.loads(self.runs_path.read_text(encoding="utf-8")) if self.runs_path.exists() else []
|
|
115
177
|
run = {"id": "pull:" + hashlib.sha256((now() + provider).encode()).hexdigest()[:12], "provider": provider, "status": "completed", "delivered": len(payload), "changed": changed, "finishedAt": now()}
|
|
178
|
+
run["translationTasks"] = len(translation_tasks)
|
|
179
|
+
if any(task["status"] != "succeeded" for task in translation_tasks):
|
|
180
|
+
run["status"] = "partial"
|
|
116
181
|
runs.append(run); self._write(self.runs_path, runs)
|
|
117
182
|
return run
|
|
118
183
|
|
|
@@ -4,10 +4,12 @@
|
|
|
4
4
|
from __future__ import annotations
|
|
5
5
|
|
|
6
6
|
import hashlib
|
|
7
|
+
import base64
|
|
8
|
+
import hmac
|
|
7
9
|
import json
|
|
8
10
|
import sqlite3
|
|
9
11
|
import uuid
|
|
10
|
-
from datetime import datetime, timezone
|
|
12
|
+
from datetime import datetime, timedelta, timezone
|
|
11
13
|
from pathlib import Path
|
|
12
14
|
from typing import Any
|
|
13
15
|
|
|
@@ -55,6 +57,20 @@ class MaggieDashStore:
|
|
|
55
57
|
created_at TEXT NOT NULL, updated_at TEXT NOT NULL,
|
|
56
58
|
UNIQUE(project_id, slug)
|
|
57
59
|
);
|
|
60
|
+
CREATE TABLE IF NOT EXISTS maggiedash_document_revisions (
|
|
61
|
+
id TEXT PRIMARY KEY, document_id TEXT NOT NULL REFERENCES maggiedash_documents(id),
|
|
62
|
+
revision INTEGER NOT NULL, title TEXT NOT NULL, slug TEXT NOT NULL,
|
|
63
|
+
locale TEXT NOT NULL, excerpt TEXT NOT NULL, content_json TEXT NOT NULL,
|
|
64
|
+
canonical_url TEXT, provenance_json TEXT NOT NULL, checksum TEXT NOT NULL,
|
|
65
|
+
created_at TEXT NOT NULL, created_by TEXT NOT NULL,
|
|
66
|
+
UNIQUE(document_id, revision)
|
|
67
|
+
);
|
|
68
|
+
CREATE TABLE IF NOT EXISTS maggiedash_redirects (
|
|
69
|
+
id TEXT PRIMARY KEY, project_id TEXT NOT NULL REFERENCES maggiedash_projects(id),
|
|
70
|
+
from_path TEXT NOT NULL, to_path TEXT NOT NULL, status_code INTEGER NOT NULL DEFAULT 301,
|
|
71
|
+
reason TEXT NOT NULL, actor_id TEXT NOT NULL, created_at TEXT NOT NULL,
|
|
72
|
+
UNIQUE(project_id, from_path)
|
|
73
|
+
);
|
|
58
74
|
CREATE TABLE IF NOT EXISTS maggiedash_approvals (
|
|
59
75
|
id TEXT PRIMARY KEY, document_id TEXT NOT NULL REFERENCES maggiedash_documents(id),
|
|
60
76
|
from_status TEXT NOT NULL, to_status TEXT NOT NULL, actor_id TEXT NOT NULL,
|
|
@@ -68,6 +84,15 @@ class MaggieDashStore:
|
|
|
68
84
|
);
|
|
69
85
|
"""
|
|
70
86
|
)
|
|
87
|
+
self._ensure_column("maggiedash_documents", "trashed_at", "TEXT")
|
|
88
|
+
self._ensure_column("maggiedash_documents", "trashed_from_status", "TEXT")
|
|
89
|
+
self._ensure_column("maggiedash_documents", "publish_at", "TEXT")
|
|
90
|
+
self.connection.commit()
|
|
91
|
+
|
|
92
|
+
def _ensure_column(self, table: str, column: str, definition: str) -> None:
|
|
93
|
+
columns = {row[1] for row in self.connection.execute(f"PRAGMA table_info({table})")}
|
|
94
|
+
if column not in columns:
|
|
95
|
+
self.connection.execute(f"ALTER TABLE {table} ADD COLUMN {column} {definition}")
|
|
71
96
|
|
|
72
97
|
def close(self) -> None:
|
|
73
98
|
self.connection.close()
|
|
@@ -97,22 +122,27 @@ class MaggieDashStore:
|
|
|
97
122
|
raise ValueError(f"invalid document status: {status}")
|
|
98
123
|
stable = {key: value for key, value in document.items() if key not in {"provenance", "checksum"}}
|
|
99
124
|
document_checksum = document.get("provenance", {}).get("checksum") or checksum(stable)
|
|
100
|
-
existing = self.connection.execute("SELECT
|
|
125
|
+
existing = self.connection.execute("SELECT * FROM maggiedash_documents WHERE id=?", (document["id"],)).fetchone()
|
|
101
126
|
if existing and existing["status"] != status:
|
|
102
127
|
raise ValueError("status changes must use transition()")
|
|
128
|
+
if existing and existing["checksum"] != document_checksum:
|
|
129
|
+
self._store_revision(existing, actor_id)
|
|
103
130
|
self.connection.execute(
|
|
104
131
|
"""INSERT INTO maggiedash_documents
|
|
105
|
-
(id,project_id,kind,title,slug,locale,status,excerpt,content_json,canonical_url,provenance_json,checksum,created_at,updated_at)
|
|
106
|
-
VALUES (
|
|
132
|
+
(id,project_id,kind,title,slug,locale,status,excerpt,content_json,canonical_url,provenance_json,checksum,created_at,updated_at,publish_at)
|
|
133
|
+
VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)
|
|
107
134
|
ON CONFLICT(id) DO UPDATE SET title=excluded.title, slug=excluded.slug,
|
|
108
135
|
locale=excluded.locale, excerpt=excluded.excerpt, content_json=excluded.content_json,
|
|
109
136
|
canonical_url=excluded.canonical_url, provenance_json=excluded.provenance_json,
|
|
110
|
-
checksum=excluded.checksum, updated_at=excluded.updated_at
|
|
137
|
+
checksum=excluded.checksum, updated_at=excluded.updated_at,
|
|
138
|
+
publish_at=excluded.publish_at, trashed_at=NULL, trashed_from_status=NULL""",
|
|
111
139
|
(document["id"], project_id, document["kind"], document["title"], document["slug"], document["locale"],
|
|
112
140
|
status, document.get("excerpt", ""), json.dumps(document["content"], ensure_ascii=False),
|
|
113
141
|
document.get("canonicalUrl"), json.dumps(document["provenance"], ensure_ascii=False), document_checksum,
|
|
114
|
-
timestamp, timestamp),
|
|
142
|
+
timestamp, timestamp, document.get("publishAt")),
|
|
115
143
|
)
|
|
144
|
+
if not existing:
|
|
145
|
+
self._store_revision(self.connection.execute("SELECT * FROM maggiedash_documents WHERE id=?", (document["id"],)).fetchone(), actor_id)
|
|
116
146
|
self._audit(project_id, "document.upserted", "document", document["id"], actor_id, "draft content stored")
|
|
117
147
|
self.connection.commit()
|
|
118
148
|
return self.get_document(project_id, document["id"]) or {}
|
|
@@ -125,8 +155,122 @@ class MaggieDashStore:
|
|
|
125
155
|
result["content"] = json.loads(result.pop("content_json"))
|
|
126
156
|
result["provenance"] = json.loads(result.pop("provenance_json"))
|
|
127
157
|
result["canonicalUrl"] = result.pop("canonical_url")
|
|
158
|
+
result["trashedAt"] = result.pop("trashed_at")
|
|
159
|
+
result["trashedFromStatus"] = result.pop("trashed_from_status")
|
|
160
|
+
result["publishAt"] = result.pop("publish_at")
|
|
128
161
|
return result
|
|
129
162
|
|
|
163
|
+
def _store_revision(self, row: sqlite3.Row, actor_id: str) -> None:
|
|
164
|
+
revision = self.connection.execute(
|
|
165
|
+
"SELECT COALESCE(MAX(revision), 0) + 1 FROM maggiedash_document_revisions WHERE document_id=?",
|
|
166
|
+
(row["id"],),
|
|
167
|
+
).fetchone()[0]
|
|
168
|
+
self.connection.execute(
|
|
169
|
+
"INSERT INTO maggiedash_document_revisions VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?)",
|
|
170
|
+
(str(uuid.uuid4()), row["id"], revision, row["title"], row["slug"], row["locale"], row["excerpt"],
|
|
171
|
+
row["content_json"], row["canonical_url"], row["provenance_json"], row["checksum"], row["updated_at"], actor_id),
|
|
172
|
+
)
|
|
173
|
+
|
|
174
|
+
def list_revisions(self, project_id: str, document_id: str) -> list[dict[str, Any]]:
|
|
175
|
+
if not self.get_document(project_id, document_id):
|
|
176
|
+
raise ValueError("document not found")
|
|
177
|
+
rows = self.connection.execute(
|
|
178
|
+
"SELECT * FROM maggiedash_document_revisions WHERE document_id=? ORDER BY revision DESC", (document_id,)
|
|
179
|
+
).fetchall()
|
|
180
|
+
return [dict(row) for row in rows]
|
|
181
|
+
|
|
182
|
+
def trash(self, project_id: str, document_id: str, actor_id: str, reason: str) -> dict[str, Any]:
|
|
183
|
+
row = self.connection.execute("SELECT status, trashed_at FROM maggiedash_documents WHERE project_id=? AND id=?", (project_id, document_id)).fetchone()
|
|
184
|
+
if not row:
|
|
185
|
+
raise ValueError("document not found")
|
|
186
|
+
if row["trashed_at"]:
|
|
187
|
+
return self.get_document(project_id, document_id) or {}
|
|
188
|
+
timestamp = now()
|
|
189
|
+
self.connection.execute("UPDATE maggiedash_documents SET trashed_at=?, trashed_from_status=?, updated_at=? WHERE id=?", (timestamp, row["status"], timestamp, document_id))
|
|
190
|
+
self._audit(project_id, "document.trashed", "document", document_id, actor_id, reason)
|
|
191
|
+
self.connection.commit()
|
|
192
|
+
return self.get_document(project_id, document_id) or {}
|
|
193
|
+
|
|
194
|
+
def restore(self, project_id: str, document_id: str, actor_id: str, reason: str) -> dict[str, Any]:
|
|
195
|
+
row = self.connection.execute("SELECT trashed_at, trashed_from_status FROM maggiedash_documents WHERE project_id=? AND id=?", (project_id, document_id)).fetchone()
|
|
196
|
+
if not row:
|
|
197
|
+
raise ValueError("document not found")
|
|
198
|
+
if not row["trashed_at"]:
|
|
199
|
+
return self.get_document(project_id, document_id) or {}
|
|
200
|
+
status = row["trashed_from_status"] if row["trashed_from_status"] in STATUSES else "draft"
|
|
201
|
+
timestamp = now()
|
|
202
|
+
self.connection.execute("UPDATE maggiedash_documents SET status=?, trashed_at=NULL, trashed_from_status=NULL, updated_at=? WHERE id=?", (status, timestamp, document_id))
|
|
203
|
+
self._audit(project_id, "document.restored", "document", document_id, actor_id, reason)
|
|
204
|
+
self.connection.commit()
|
|
205
|
+
return self.get_document(project_id, document_id) or {}
|
|
206
|
+
|
|
207
|
+
def schedule_publish(self, project_id: str, document_id: str, publish_at: str, actor_id: str, reason: str) -> dict[str, Any]:
|
|
208
|
+
try:
|
|
209
|
+
target = datetime.fromisoformat(publish_at.replace("Z", "+00:00"))
|
|
210
|
+
except ValueError as exc:
|
|
211
|
+
raise ValueError("publish_at must be an ISO-8601 timestamp") from exc
|
|
212
|
+
if target.tzinfo is None:
|
|
213
|
+
raise ValueError("publish_at must include a timezone")
|
|
214
|
+
document = self.get_document(project_id, document_id)
|
|
215
|
+
if not document:
|
|
216
|
+
raise ValueError("document not found")
|
|
217
|
+
if document["status"] not in {"approved", "scheduled"}:
|
|
218
|
+
raise ValueError("only approved documents can be scheduled")
|
|
219
|
+
if document["status"] == "approved":
|
|
220
|
+
self.transition(project_id, document_id, "scheduled", actor_id, reason)
|
|
221
|
+
self.connection.execute("UPDATE maggiedash_documents SET publish_at=?, updated_at=? WHERE id=?", (publish_at, now(), document_id))
|
|
222
|
+
self._audit(project_id, "document.publish_scheduled", "document", document_id, actor_id, reason)
|
|
223
|
+
self.connection.commit()
|
|
224
|
+
return self.get_document(project_id, document_id) or {}
|
|
225
|
+
|
|
226
|
+
def duplicate(self, project_id: str, document_id: str, new_id: str, new_slug: str, actor_id: str, reason: str) -> dict[str, Any]:
|
|
227
|
+
source = self.get_document(project_id, document_id)
|
|
228
|
+
if not source:
|
|
229
|
+
raise ValueError("document not found")
|
|
230
|
+
if self.connection.execute("SELECT 1 FROM maggiedash_documents WHERE project_id=? AND slug=?", (project_id, new_slug)).fetchone():
|
|
231
|
+
raise ValueError("new slug already exists")
|
|
232
|
+
copy = dict(source)
|
|
233
|
+
copy.update({"id": new_id, "slug": new_slug, "status": "draft", "provenance": {**source["provenance"], "duplicatedFrom": document_id}})
|
|
234
|
+
copy.pop("trashedAt", None); copy.pop("trashedFromStatus", None); copy.pop("publishAt", None)
|
|
235
|
+
result = self.put_document(project_id, copy, actor_id)
|
|
236
|
+
self._audit(project_id, "document.duplicated", "document", new_id, actor_id, reason)
|
|
237
|
+
self.connection.commit()
|
|
238
|
+
return result
|
|
239
|
+
|
|
240
|
+
def add_redirect(self, project_id: str, from_path: str, to_path: str, actor_id: str, reason: str, status_code: int = 301) -> dict[str, Any]:
|
|
241
|
+
if not from_path.startswith("/") or not to_path.startswith("/") or from_path == to_path:
|
|
242
|
+
raise ValueError("redirect paths must be distinct absolute paths")
|
|
243
|
+
if status_code not in {301, 302, 307, 308}:
|
|
244
|
+
raise ValueError("status_code must be 301, 302, 307 or 308")
|
|
245
|
+
record = (str(uuid.uuid4()), project_id, from_path, to_path, status_code, reason, actor_id, now())
|
|
246
|
+
self.connection.execute("INSERT INTO maggiedash_redirects VALUES (?,?,?,?,?,?,?,?) ON CONFLICT(project_id,from_path) DO UPDATE SET to_path=excluded.to_path, status_code=excluded.status_code, reason=excluded.reason, actor_id=excluded.actor_id, created_at=excluded.created_at", record)
|
|
247
|
+
self._audit(project_id, "redirect.upserted", "redirect", from_path, actor_id, reason)
|
|
248
|
+
self.connection.commit()
|
|
249
|
+
row = self.connection.execute("SELECT * FROM maggiedash_redirects WHERE project_id=? AND from_path=?", (project_id, from_path)).fetchone()
|
|
250
|
+
return dict(row)
|
|
251
|
+
|
|
252
|
+
def issue_preview(self, project_id: str, document_id: str, secret: str, ttl_seconds: int = 900) -> dict[str, Any]:
|
|
253
|
+
if not secret:
|
|
254
|
+
raise ValueError("preview secret is required")
|
|
255
|
+
if not self.get_document(project_id, document_id):
|
|
256
|
+
raise ValueError("document not found")
|
|
257
|
+
expires = int((datetime.now(timezone.utc) + timedelta(seconds=ttl_seconds)).timestamp())
|
|
258
|
+
payload = f"{project_id}:{document_id}:{expires}".encode()
|
|
259
|
+
encoded = base64.urlsafe_b64encode(payload).decode().rstrip("=")
|
|
260
|
+
signature = hmac.new(secret.encode(), encoded.encode(), hashlib.sha256).hexdigest()
|
|
261
|
+
return {"token": f"{encoded}.{signature}", "expiresAt": datetime.fromtimestamp(expires, timezone.utc).isoformat().replace("+00:00", "Z")}
|
|
262
|
+
|
|
263
|
+
@staticmethod
|
|
264
|
+
def verify_preview(token: str, secret: str, project_id: str, document_id: str) -> bool:
|
|
265
|
+
try:
|
|
266
|
+
encoded, signature = token.split(".", 1)
|
|
267
|
+
expected = hmac.new(secret.encode(), encoded.encode(), hashlib.sha256).hexdigest()
|
|
268
|
+
payload = base64.urlsafe_b64decode(encoded + "=" * (-len(encoded) % 4)).decode()
|
|
269
|
+
token_project, token_document, expires = payload.split(":", 2)
|
|
270
|
+
return hmac.compare_digest(signature, expected) and token_project == project_id and token_document == document_id and int(expires) >= int(datetime.now(timezone.utc).timestamp())
|
|
271
|
+
except (ValueError, TypeError, UnicodeDecodeError):
|
|
272
|
+
return False
|
|
273
|
+
|
|
130
274
|
def transition(self, project_id: str, document_id: str, target: str, actor_id: str, reason: str) -> dict[str, Any]:
|
|
131
275
|
if target not in STATUSES:
|
|
132
276
|
raise ValueError(f"invalid target status: {target}")
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
"""Validate the provider-neutral MaggieDash dashboard chrome contract."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import Iterable
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
SCHEMA = "maggiedash-dashboard-ui.v1"
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def _component_source(name: str, source: str) -> str:
|
|
14
|
+
match = re.search(rf"(?:function\s+{re.escape(name)}\s*\([^)]*\)|const\s+{re.escape(name)}\s*=\s*\([^)]*\))(?P<body>[\s\S]{{0,12000}})", source)
|
|
15
|
+
return match.group(0) if match else ""
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def validate(contract: object, sources: Iterable[str] = ()) -> dict:
|
|
19
|
+
errors: list[str] = []
|
|
20
|
+
if not isinstance(contract, dict):
|
|
21
|
+
return {"schemaVersion": SCHEMA, "passed": False, "errors": ["contract must be an object"]}
|
|
22
|
+
if contract.get("schemaVersion") != SCHEMA:
|
|
23
|
+
errors.append(f"schemaVersion must be {SCHEMA}")
|
|
24
|
+
shell = contract.get("shell")
|
|
25
|
+
if not isinstance(shell, dict):
|
|
26
|
+
errors.append("shell is required")
|
|
27
|
+
elif shell.get("reference") != "users-workspace" or shell.get("navigation") != "sidebar-plus-content-tabs":
|
|
28
|
+
errors.append("shell must declare the users-workspace reference and sidebar-plus-content-tabs navigation")
|
|
29
|
+
components = contract.get("components")
|
|
30
|
+
if not isinstance(components, list) or not components:
|
|
31
|
+
errors.append("components must be a non-empty list")
|
|
32
|
+
components = []
|
|
33
|
+
source = "\n".join(str(item) for item in sources)
|
|
34
|
+
for component in components:
|
|
35
|
+
if not isinstance(component, dict) or not component.get("name"):
|
|
36
|
+
errors.append("each component needs a name")
|
|
37
|
+
continue
|
|
38
|
+
name = str(component["name"])
|
|
39
|
+
component_source = _component_source(name, source)
|
|
40
|
+
if source and not component_source:
|
|
41
|
+
errors.append(f"component source is missing: {name}")
|
|
42
|
+
continue
|
|
43
|
+
props = [str(prop) for prop in component.get("props", [])]
|
|
44
|
+
forbidden = [str(prop) for prop in component.get("forbiddenProps", [])]
|
|
45
|
+
if component_source:
|
|
46
|
+
signature = component_source.split("=>", 1)[0] if "=>" in component_source else component_source[:500]
|
|
47
|
+
for prop in props:
|
|
48
|
+
if len(re.findall(rf"\b{re.escape(prop)}\b", component_source)) < 2:
|
|
49
|
+
errors.append(f"{name}.{prop} is declared but not evidenced in rendered output")
|
|
50
|
+
for prop in forbidden:
|
|
51
|
+
if re.search(rf"\b{re.escape(prop)}\b", signature):
|
|
52
|
+
errors.append(f"{name} declares forbidden dead prop: {prop}")
|
|
53
|
+
return {"schemaVersion": SCHEMA, "passed": not errors, "errors": errors, "components": [item.get("name") for item in components if isinstance(item, dict)]}
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def load_and_validate(contract_path: Path, source_paths: Iterable[Path] = ()) -> dict:
|
|
57
|
+
import json
|
|
58
|
+
contract = json.loads(contract_path.read_text(encoding="utf-8"))
|
|
59
|
+
sources = [path.read_text(encoding="utf-8", errors="replace") for path in source_paths]
|
|
60
|
+
return validate(contract, sources)
|
|
@@ -171,12 +171,17 @@ def _active(item: dict[str, Any], project_id: str = "") -> bool:
|
|
|
171
171
|
return not expires or expires > now()
|
|
172
172
|
|
|
173
173
|
|
|
174
|
-
def relevant(project: str | Path, *, skill: str = "", query: str = "", project_id: str = "") -> dict[str, list[dict[str, Any]]]:
|
|
174
|
+
def relevant(project: str | Path, *, skill: str = "", query: str = "", project_id: str = "", status: str | None = None) -> dict[str, list[dict[str, Any]]]:
|
|
175
175
|
terms = set(re.findall(r"[a-z0-9_-]+", f"{skill} {query}".lower()))
|
|
176
176
|
result: dict[str, list[dict[str, Any]]] = {kind: [] for kind in KINDS}
|
|
177
177
|
for kind in KINDS:
|
|
178
178
|
for item in read(project, kind)["items"]:
|
|
179
|
-
if
|
|
179
|
+
if status is not None:
|
|
180
|
+
if item.get("status") != status:
|
|
181
|
+
continue
|
|
182
|
+
if item.get("scope") == "project" and item.get("projectId") not in {project_id or root(project).name, ""}:
|
|
183
|
+
continue
|
|
184
|
+
elif kind != "errors" and not _active(item, project_id or root(project).name):
|
|
180
185
|
continue
|
|
181
186
|
haystack = " ".join(str(item.get(key, "")) for key in ("type", "title", "rule", "problem", "solution", "prevention", "skill", "message", "fingerprint", "triggers")).lower()
|
|
182
187
|
if not terms or terms & set(re.findall(r"[a-z0-9_-]+", haystack)):
|
|
@@ -55,6 +55,12 @@ def filter_routes(routes: list[dict[str, str]], origin: str, content_types: set[
|
|
|
55
55
|
reason = "content type not enabled"
|
|
56
56
|
elif not absolute_url(url, origin):
|
|
57
57
|
reason = "URL must be an absolute HTTP(S) URL on the approved origin"
|
|
58
|
+
elif route.get("indexable") is False:
|
|
59
|
+
reason = "route is explicitly non-indexable"
|
|
60
|
+
elif route.get("canonicalUrl") and route.get("canonicalUrl") != url:
|
|
61
|
+
reason = "route canonical URL does not self-canonicalize"
|
|
62
|
+
elif route.get("searchable") is False:
|
|
63
|
+
reason = "route is explicitly not searchable"
|
|
58
64
|
elif EXCLUDED_PATHS.search(parsed.path) or any(key in parse_qs(parsed.query) for key in ("q", "search", "filter", "page")):
|
|
59
65
|
reason = "search, filter, or non-canonical route"
|
|
60
66
|
elif route.get("type", "").lower() in {"rss", "atom", "feed"}:
|
|
@@ -86,12 +92,12 @@ def build_plan(routes: list[dict[str, str]], origin: str, content_types: set[str
|
|
|
86
92
|
for content_type in sorted(content_types):
|
|
87
93
|
chunks = []
|
|
88
94
|
entries = groups.get(content_type, []) or []
|
|
89
|
-
for index in range(0,
|
|
95
|
+
for index in range(0, len(entries), chunk_target):
|
|
90
96
|
chunk_entries = entries[index:index + chunk_target]
|
|
91
97
|
filename = f"sitemap-{content_type}-{index // chunk_target + 1}.xml"
|
|
92
98
|
xml = xml_file(chunk_entries)
|
|
93
99
|
url = origin.rstrip("/") + "/" + filename
|
|
94
|
-
chunks.append({"filename": filename, "url": url, "entries": len(chunk_entries), "bytes": len(xml.encode()), "sha256": hashlib.sha256(xml.encode()).hexdigest(), "xml": xml})
|
|
100
|
+
chunks.append({"filename": filename, "url": url, "entries": len(chunk_entries), "routes": chunk_entries, "bytes": len(xml.encode()), "sha256": hashlib.sha256(xml.encode()).hexdigest(), "xml": xml})
|
|
95
101
|
sitemap_urls.append(url)
|
|
96
102
|
group_plans.append({"contentType": content_type, "chunks": chunks})
|
|
97
103
|
index_xml = '<?xml version="1.0" encoding="UTF-8"?>\n<sitemapindex xmlns="http://www.sitemaps.org/schemas/sitemap/0.9">' + "".join(f"<sitemap><loc>{escape(url)}</loc></sitemap>" for url in sitemap_urls) + "</sitemapindex>\n"
|
|
@@ -103,7 +109,7 @@ def build_plan(routes: list[dict[str, str]], origin: str, content_types: set[str
|
|
|
103
109
|
return plan
|
|
104
110
|
|
|
105
111
|
|
|
106
|
-
def validate_plan_data(plan: dict) -> dict:
|
|
112
|
+
def validate_plan_data(plan: dict, strict_semantic: bool = False) -> dict:
|
|
107
113
|
errors = []
|
|
108
114
|
warnings = []
|
|
109
115
|
origin = plan.get("origin", "")
|
|
@@ -117,6 +123,21 @@ def validate_plan_data(plan: dict) -> dict:
|
|
|
117
123
|
errors.append(f"chunk exceeds URL limit: {chunk.get('filename')}")
|
|
118
124
|
if chunk.get("bytes", 0) > HARD_BYTES_LIMIT:
|
|
119
125
|
errors.append(f"chunk exceeds byte limit: {chunk.get('filename')}")
|
|
126
|
+
for route in chunk.get("routes", []):
|
|
127
|
+
if route.get("indexable") is False or route.get("searchable") is False:
|
|
128
|
+
errors.append(f"non-indexable/searchable route was emitted: {route.get('url')}")
|
|
129
|
+
if route.get("canonicalUrl") and route.get("canonicalUrl") != route.get("url"):
|
|
130
|
+
errors.append(f"non-self-canonical route was emitted: {route.get('url')}")
|
|
131
|
+
if "contentWords" in route and int(route.get("contentWords") or 0) < 1:
|
|
132
|
+
errors.append(f"route has no searchable content evidence: {route.get('url')}")
|
|
133
|
+
if "title" in route and not str(route.get("title") or "").strip():
|
|
134
|
+
errors.append(f"route has no title evidence: {route.get('url')}")
|
|
135
|
+
if route.get("lastmod") and route.get("lastmodSource") in {"updated_at", "sync", "pull", "deploy"}:
|
|
136
|
+
errors.append(f"lastmod is derived from an operational timestamp: {route.get('url')}")
|
|
137
|
+
if route.get("lastmod") and route.get("lastmodSource") not in {None, "content-change", "source-revision"}:
|
|
138
|
+
errors.append(f"lastmod provenance is not a content change: {route.get('url')}")
|
|
139
|
+
if route.get("lastmod") and strict_semantic and route.get("lastmodSource") not in {"content-change", "source-revision"}:
|
|
140
|
+
errors.append(f"lastmod evidence is required for strict semantic validation: {route.get('url')}")
|
|
120
141
|
try:
|
|
121
142
|
root = ElementTree.fromstring(chunk.get("xml", ""))
|
|
122
143
|
locs = [element.text or "" for element in root.iter() if element.tag.endswith("loc")]
|
|
@@ -140,5 +161,30 @@ def validate_plan_data(plan: dict) -> dict:
|
|
|
140
161
|
except ElementTree.ParseError:
|
|
141
162
|
errors.append("sitemap index is not valid XML")
|
|
142
163
|
if not expected_urls:
|
|
143
|
-
warnings.append("no sitemap chunks generated")
|
|
164
|
+
warnings.append("no sitemap chunks generated; empty content types are not advertised")
|
|
165
|
+
if strict_semantic:
|
|
166
|
+
for group in plan.get("groups", []):
|
|
167
|
+
for chunk in group.get("chunks", []):
|
|
168
|
+
for route in chunk.get("routes", []):
|
|
169
|
+
if not str(route.get("title") or "").strip(): warnings.append(f"semantic title evidence missing: {route.get('url')}")
|
|
170
|
+
if "contentWords" not in route: warnings.append(f"semantic content evidence missing: {route.get('url')}")
|
|
171
|
+
if warnings:
|
|
172
|
+
errors.extend(warnings)
|
|
144
173
|
return {"status": "pass" if not errors else "fail", "errors": errors, "warnings": warnings}
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def agent_files(routes: list[dict[str, str]], origin: str, locale: str = "en") -> dict[str, str]:
|
|
177
|
+
"""Generate compact, locale-aware agent discovery documents."""
|
|
178
|
+
visible = [route for route in routes if route.get("indexable") is not False and route.get("searchable") is not False and absolute_url(route.get("url", ""), origin)]
|
|
179
|
+
visible.sort(key=lambda route: route.get("url", ""))
|
|
180
|
+
lines = [f"# {origin} ({locale})", "", "This file lists indexable public routes for agent discovery.", "", "## Routes"]
|
|
181
|
+
for route in visible:
|
|
182
|
+
label = route.get("title") or route.get("url", "").rstrip("/").rsplit("/", 1)[-1] or "home"
|
|
183
|
+
lines.append(f"- [{label}]({route['url']})")
|
|
184
|
+
sitemap = [f"# Sitemap ({locale})", "", f"- XML sitemap: {origin.rstrip('/')}/sitemap.xml", "", "## Public routes"]
|
|
185
|
+
sitemap.extend(f"- {route['url']}" for route in visible)
|
|
186
|
+
insights = [f"# Content insights ({locale})", "", f"- Indexable routes: {len(visible)}", f"- Content types: {', '.join(sorted({str(route.get('type', 'unknown')) for route in visible})) or 'none'}", "", "## Freshness"]
|
|
187
|
+
for route in visible:
|
|
188
|
+
if route.get("lastmod"):
|
|
189
|
+
insights.append(f"- {route['url']}: last content change {route['lastmod']}")
|
|
190
|
+
return {"llms.txt": "\n".join(lines) + "\n", "sitemap.md": "\n".join(sitemap) + "\n", "insights.md": "\n".join(insights) + "\n"}
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
"""Resolve route components from source imports, never from sibling basenames."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import hashlib
|
|
6
|
+
import re
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
IMPORT_RE = re.compile(r"(?:import\s+(?:[^;\n]*?\s+from\s+)?|export\s+[^;\n]*?\s+from\s+)[\"']([^\"']+)[\"']")
|
|
11
|
+
EXTENSIONS = ("", ".astro", ".tsx", ".jsx", ".ts", ".js", ".vue", ".svelte", ".html")
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def resolve_import(route: Path, specifier: str) -> Path | None:
|
|
15
|
+
if not specifier.startswith("."):
|
|
16
|
+
return None
|
|
17
|
+
base = (route.parent / specifier).resolve()
|
|
18
|
+
candidates = [base]
|
|
19
|
+
candidates.extend(Path(str(base) + extension) for extension in EXTENSIONS[1:])
|
|
20
|
+
candidates.extend(base / f"index{extension}" for extension in EXTENSIONS)
|
|
21
|
+
for candidate in candidates:
|
|
22
|
+
if candidate.is_file():
|
|
23
|
+
return candidate
|
|
24
|
+
return None
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def imported_components(route: Path) -> list[dict[str, str | None]]:
|
|
28
|
+
"""Return only source imports that resolve inside the project.
|
|
29
|
+
|
|
30
|
+
Bare package imports are intentionally recorded without a local path; they
|
|
31
|
+
are dependencies, not route-local components. The route source remains
|
|
32
|
+
the authority, so a same-named sibling file is never selected implicitly.
|
|
33
|
+
"""
|
|
34
|
+
try:
|
|
35
|
+
source = route.read_text(encoding="utf-8", errors="replace")
|
|
36
|
+
except OSError:
|
|
37
|
+
return []
|
|
38
|
+
result = []
|
|
39
|
+
seen = set()
|
|
40
|
+
for specifier in IMPORT_RE.findall(source):
|
|
41
|
+
resolved = resolve_import(route, specifier)
|
|
42
|
+
key = (specifier, str(resolved) if resolved else None)
|
|
43
|
+
if key in seen:
|
|
44
|
+
continue
|
|
45
|
+
seen.add(key)
|
|
46
|
+
result.append({
|
|
47
|
+
"specifier": specifier,
|
|
48
|
+
"path": str(resolved) if resolved else None,
|
|
49
|
+
"fingerprint": hashlib.sha256(resolved.read_bytes()).hexdigest() if resolved else None,
|
|
50
|
+
})
|
|
51
|
+
return result
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
"""Validation for explicit, sanitized development fixture evidence."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
SECRET = re.compile(r"(?:secret|token|password|api[_-]?key|database_url)", re.I)
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def validate_manifest(value: object) -> dict:
|
|
12
|
+
errors = []
|
|
13
|
+
if not isinstance(value, dict):
|
|
14
|
+
return {"schemaVersion": "maggie-seed-manifest.v1", "passed": False, "errors": ["manifest must be an object"]}
|
|
15
|
+
if value.get("schemaVersion") != "maggie-seed-manifest.v1": errors.append("unsupported schemaVersion")
|
|
16
|
+
if value.get("optIn") is not True: errors.append("optIn must be true")
|
|
17
|
+
if value.get("sanitized") is not True: errors.append("sanitized must be true")
|
|
18
|
+
fixtures = value.get("fixtures")
|
|
19
|
+
if not isinstance(fixtures, list) or not fixtures: errors.append("fixtures must be a non-empty list")
|
|
20
|
+
for item in fixtures if isinstance(fixtures, list) else []:
|
|
21
|
+
if not isinstance(item, dict) or not item.get("name") or not item.get("rows"):
|
|
22
|
+
errors.append("each fixture requires name and non-empty rows")
|
|
23
|
+
if SECRET.search(str(item)):
|
|
24
|
+
errors.append("fixture contains a secret-like field")
|
|
25
|
+
return {"schemaVersion": "maggie-seed-manifest.v1", "passed": not errors, "errors": errors, "fixtureCount": len(fixtures) if isinstance(fixtures, list) else 0}
|