emdash-plugin-sitegraph 0.0.0-stage → 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.mjs ADDED
@@ -0,0 +1,957 @@
1
+ import { PluginRouteError, definePlugin } from "emdash";
2
+ import { z } from "astro/zod";
3
+
4
+ //#region src/domain/graph.ts
5
+ const NODE_TYPES = [
6
+ "CONTENT",
7
+ "URL",
8
+ "FORM",
9
+ "SERVICE",
10
+ "WORKFLOW",
11
+ "TEAM_MEMBER"
12
+ ];
13
+ /** Node types a person can create; CONTENT and URL only come from scans. */
14
+ const DOCUMENTED_NODE_TYPES = [
15
+ "FORM",
16
+ "SERVICE",
17
+ "WORKFLOW",
18
+ "TEAM_MEMBER"
19
+ ];
20
+ /** Relations a person can document; PUBLISHES_AS and LINKS_TO only come from scans. */
21
+ const DOCUMENTED_RELATION_TYPES = [
22
+ "SUBMITS_TO",
23
+ "DEPENDS_ON",
24
+ "PART_OF",
25
+ "OWNED_BY",
26
+ "RELATED_TO"
27
+ ];
28
+ const CRITICALITIES = [
29
+ "LOW",
30
+ "MEDIUM",
31
+ "HIGH",
32
+ "CRITICAL"
33
+ ];
34
+ const SCHEMA_VERSION = 1;
35
+ const contentNodeId = (collection, entryId) => `content:${collection}:${entryId}`;
36
+ const urlNodeId = (path) => `url:${path}`;
37
+ const edgeId = (source, relation, target, fieldPath = "") => `${source}|${relation}|${target}|${fieldPath}`;
38
+
39
+ //#endregion
40
+ //#region src/domain/impact.ts
41
+ const MAX_DEPTH$1 = 3;
42
+ const MAX_NODES = 500;
43
+ const MAX_EDGES = 1e3;
44
+ /**
45
+ * Most relations read "source depends on target" (a page links to a URL, a form submits to a
46
+ * CRM). PART_OF reads the other way: the workflow depends on its parts. RELATED_TO has no
47
+ * direction. So "inbound" (what depends on this) walks edges backwards, except PART_OF.
48
+ */
49
+ function follows(relation, dir, direction) {
50
+ if (relation === "RELATED_TO") return true;
51
+ return (relation === "PART_OF" ? dir === "inbound" ? "outbound" : "inbound" : dir) === direction;
52
+ }
53
+ /**
54
+ * Bounded breadth-first search. "inbound" finds what depends on the start node (what could
55
+ * break if it changes); "outbound" finds what it depends on. Each node is
56
+ * reached once, by its shortest path, so cycles end the walk. Reachability is potential
57
+ * impact, never proof of it.
58
+ */
59
+ async function impact(start, direction, depth, fetchEdges) {
60
+ const maxDepth = Math.max(1, Math.min(MAX_DEPTH$1, Math.floor(depth)));
61
+ const reached = new Map([[start, {
62
+ nodeId: start,
63
+ depth: 0,
64
+ path: [],
65
+ confirmed: true
66
+ }]]);
67
+ const edges = /* @__PURE__ */ new Map();
68
+ let frontier = [start];
69
+ let truncated = false;
70
+ for (let level = 1; level <= maxDepth && frontier.length > 0 && !truncated; level++) {
71
+ const next = [];
72
+ let batch = frontier;
73
+ while (batch.length > 0 && !truncated) {
74
+ const samePage = [];
75
+ for (const dir of ["outbound", "inbound"]) for (const edge of await fetchEdges(batch, dir)) {
76
+ const free = edge.relation === "PUBLISHES_AS";
77
+ if (direction !== "both" && !free && !follows(edge.relation, dir, direction)) continue;
78
+ const from = dir === "outbound" ? edge.sourceNodeId : edge.targetNodeId;
79
+ const to = dir === "outbound" ? edge.targetNodeId : edge.sourceNodeId;
80
+ const parent = reached.get(from);
81
+ if (!parent) continue;
82
+ if (edges.size >= MAX_EDGES) {
83
+ truncated = true;
84
+ break;
85
+ }
86
+ edges.set(edge.id, edge);
87
+ if (reached.has(to)) continue;
88
+ if (reached.size >= MAX_NODES) {
89
+ truncated = true;
90
+ break;
91
+ }
92
+ reached.set(to, {
93
+ nodeId: to,
94
+ depth: free ? parent.depth : level,
95
+ path: [...parent.path, edge],
96
+ confirmed: parent.confirmed && edge.provenance !== "INFERRED"
97
+ });
98
+ (free ? samePage : next).push(to);
99
+ }
100
+ batch = samePage;
101
+ }
102
+ frontier = next;
103
+ }
104
+ reached.delete(start);
105
+ return {
106
+ start,
107
+ hits: [...reached.values()],
108
+ edges: [...edges.values()],
109
+ truncated
110
+ };
111
+ }
112
+
113
+ //#endregion
114
+ //#region src/domain/links.ts
115
+ const MAX_DEPTH = 32;
116
+ const UNSAFE_SCHEME = /^\s*(javascript|data|vbscript|file|mailto|tel):/i;
117
+ /**
118
+ * Find links in an entry's data: Portable Text link marks (`{_type: "link", href}`)
119
+ * anywhere in the tree, plus the values of the given `url`-type fields.
120
+ * Plain strings that merely look like URLs are ignored on purpose.
121
+ */
122
+ function extractLinks(data, urlFields = []) {
123
+ const found = [];
124
+ for (const field of urlFields) {
125
+ const value = data[field];
126
+ if (typeof value === "string" && value.trim()) found.push({
127
+ href: value.trim(),
128
+ field,
129
+ path: field
130
+ });
131
+ }
132
+ const walk = (value, field, path, depth) => {
133
+ if (depth > MAX_DEPTH || value === null || typeof value !== "object") return;
134
+ if (Array.isArray(value)) {
135
+ value.forEach((item, i) => walk(item, field, `${path}[${i}]`, depth + 1));
136
+ return;
137
+ }
138
+ const obj = value;
139
+ if (obj._type === "link" && typeof obj.href === "string" && obj.href.trim()) found.push({
140
+ href: obj.href.trim(),
141
+ field,
142
+ path
143
+ });
144
+ for (const [key, child] of Object.entries(obj)) walk(child, field, `${path}.${key}`, depth + 1);
145
+ };
146
+ for (const [field, value] of Object.entries(data)) walk(value, field, field, 0);
147
+ return found;
148
+ }
149
+ /**
150
+ * Turn an href into the site-relative path that identifies a page, or null when the
151
+ * link is external, unsafe or not a page link. Fragments and query strings are dropped
152
+ * (`/pricing?plan=pro` is the pricing page; the raw href stays in the edge evidence) and
153
+ * trailing slashes are ignored so "/a/" and "/a" are the same page.
154
+ */
155
+ function internalPath(href, sourcePath, siteUrl) {
156
+ if (UNSAFE_SCHEME.test(href) || href.trim().startsWith("#")) return null;
157
+ const origin = siteUrl ? new URL(siteUrl).origin : "http://sitegraph.invalid";
158
+ let url;
159
+ try {
160
+ url = new URL(href, new URL(sourcePath || "/", origin));
161
+ } catch {
162
+ return null;
163
+ }
164
+ if (url.protocol !== "http:" && url.protocol !== "https:") return null;
165
+ if (url.origin !== origin) return null;
166
+ let path = url.pathname.replace(/\/{2,}/g, "/");
167
+ if (path.length > 1 && path.endsWith("/")) path = path.slice(0, -1);
168
+ return path;
169
+ }
170
+ /**
171
+ * The public path of an entry from its collection's URL pattern, mirroring EmDash's
172
+ * `interpolateUrlPattern` for `{slug}` and `{id}`. Returns null for patterns using
173
+ * tokens we don't resolve (dates, locales), so we never invent a wrong URL.
174
+ */
175
+ function entryPath(pattern, collection, slug, id) {
176
+ let path = (pattern ?? `/${encodeURIComponent(collection)}/{slug}`).replaceAll("{slug}", encodeURIComponent(slug)).replaceAll("{id}", encodeURIComponent(id));
177
+ if (path.includes("{")) return null;
178
+ path = path.replace(/\/{2,}/g, "/");
179
+ if (!path.startsWith("/")) path = `/${path}`;
180
+ if (path.length > 1 && path.endsWith("/")) path = path.slice(0, -1);
181
+ return path;
182
+ }
183
+ /**
184
+ * A matcher for paths an entry of this collection could live at, or null when the pattern
185
+ * uses tokens we don't resolve. Used to tell a broken link (looks like an entry, none is
186
+ * there) from a link to some other page of the site (home, listings, feeds).
187
+ */
188
+ function entryPathMatcher(pattern, collection) {
189
+ let base = pattern ?? `/${encodeURIComponent(collection)}/{slug}`;
190
+ if (/\{(?!slug\}|id\})[^}]*\}/.test(base)) return null;
191
+ base = base.replace(/\/{2,}/g, "/");
192
+ if (!base.startsWith("/")) base = `/${base}`;
193
+ if (base.length > 1 && base.endsWith("/")) base = base.slice(0, -1);
194
+ const source = base.split(/(\{slug\}|\{id\})/).map((part) => part === "{slug}" || part === "{id}" ? "[^/]+" : part.replace(/[.*+?^${}()|[\]\\]/g, "\\$&")).join("");
195
+ return new RegExp(`^${source}$`);
196
+ }
197
+
198
+ //#endregion
199
+ //#region src/store.ts
200
+ const STORAGE = {
201
+ nodes: { indexes: [
202
+ "type",
203
+ "active",
204
+ "search",
205
+ "provenance",
206
+ "resolved"
207
+ ] },
208
+ edges: { indexes: [
209
+ "sourceNodeId",
210
+ "targetNodeId",
211
+ "provenance",
212
+ "active"
213
+ ] }
214
+ };
215
+ const nodesOf = (ctx) => ctx.storage.nodes;
216
+ const edgesOf = (ctx) => ctx.storage.edges;
217
+ const PAGE = 100;
218
+ /** Documents matching `where`, page by page, stopping once `max` are read. */
219
+ async function queryAll(collection, where, max = Infinity) {
220
+ const out = [];
221
+ let cursor;
222
+ do {
223
+ const page = await collection.query({
224
+ where,
225
+ limit: PAGE,
226
+ cursor
227
+ });
228
+ out.push(...page.items);
229
+ cursor = page.hasMore && out.length < max ? page.cursor : void 0;
230
+ } while (cursor);
231
+ return out.slice(0, max);
232
+ }
233
+ /**
234
+ * Active edges touching any of `nodeIds`, used by impact and neighbourhood views. Stops reading
235
+ * at `max`, so a hub page linked from every footer can't make a request read the whole table.
236
+ */
237
+ async function activeEdges(ctx, nodeIds, direction, max = Infinity) {
238
+ const field = direction === "outbound" ? "sourceNodeId" : "targetNodeId";
239
+ const out = [];
240
+ for (let i = 0; i < nodeIds.length && out.length < max; i += PAGE) {
241
+ const rows = await queryAll(edgesOf(ctx), {
242
+ [field]: { in: nodeIds.slice(i, i + PAGE) },
243
+ active: true
244
+ }, max - out.length);
245
+ for (const row of rows) out.push({
246
+ id: row.id,
247
+ ...row.data
248
+ });
249
+ }
250
+ return out;
251
+ }
252
+
253
+ //#endregion
254
+ //#region src/scan.ts
255
+ const SCAN_KEY = "scan";
256
+ const LAST_SCAN_KEY = "lastScan";
257
+ const BATCH = 25;
258
+ const STALE_AFTER_MS = 120 * 1e3;
259
+ const MAX_ERRORS = 20;
260
+ const errorText = (error) => error instanceof Error ? error.message : String(error);
261
+ async function collectionPlans(ctx) {
262
+ return (await ctx.schema?.listCollections() ?? []).map((c) => ({
263
+ slug: c.slug,
264
+ label: c.labelSingular ?? c.label,
265
+ pattern: c.urlPattern,
266
+ routable: c.routable,
267
+ titleField: c.titleField,
268
+ urlFields: c.fields.filter((f) => f.type === "url").map((f) => f.slug)
269
+ }));
270
+ }
271
+ /**
272
+ * Could an entry live at this path? Yes when it fits a routable collection's URL pattern and
273
+ * doesn't look like a file. A link to such a path with no entry behind it is broken; a link
274
+ * to anything else (home page, listings, feeds) points at a page we simply don't map.
275
+ */
276
+ function looksLikeEntryPath(path, plans) {
277
+ if (path.slice(path.lastIndexOf("/") + 1).includes(".")) return false;
278
+ return plans.some((p) => p.routable && entryPathMatcher(p.pattern, p.slug)?.test(path));
279
+ }
280
+ function entryLabel(item, plan) {
281
+ const title = plan.titleField ? item.data[plan.titleField] : item.data.title;
282
+ if (typeof title === "string" && title.trim()) return title.trim().slice(0, 200);
283
+ return item.slug ?? item.id;
284
+ }
285
+ /** A discovered node merged over what's stored, so human annotations and firstSeenAt survive. */
286
+ function discoveredNode(existing, fresh, now, scanId) {
287
+ return {
288
+ ...existing,
289
+ ...fresh,
290
+ search: fresh.label.toLowerCase(),
291
+ provenance: "DISCOVERED",
292
+ active: true,
293
+ firstSeenAt: existing?.firstSeenAt ?? now,
294
+ lastSeenAt: now,
295
+ schemaVersion: SCHEMA_VERSION,
296
+ scanId
297
+ };
298
+ }
299
+ /**
300
+ * Recompute a URL node's status from the edges that point at it, the single source of truth:
301
+ * a published entry there → resolved; otherwise broken if it looks like an entry path, or
302
+ * unknown if not; and with nothing pointing at it at all, it leaves the map.
303
+ */
304
+ async function settleUrl(ctx, plans, nodeId) {
305
+ const nodes = nodesOf(ctx);
306
+ const node = await nodes.get(nodeId);
307
+ if (!node || node.type !== "URL" || node.provenance !== "DISCOVERED") return false;
308
+ const inbound = await activeEdges(ctx, [nodeId], "inbound", 50);
309
+ const hasPage = inbound.some((e) => e.relation === "PUBLISHES_AS");
310
+ const next = { ...node };
311
+ if (!hasPage && inbound.length === 0) next.active = false;
312
+ else if (hasPage) next.resolved = true;
313
+ else if (looksLikeEntryPath(node.ref, plans)) next.resolved = false;
314
+ else delete next.resolved;
315
+ if (next.active === node.active && next.resolved === node.resolved) return false;
316
+ await nodes.put(nodeId, next);
317
+ return true;
318
+ }
319
+ /**
320
+ * Re-read one entry's graph: its CONTENT node, its URL, and its outgoing links. Outgoing
321
+ * discovered edges it no longer has are retired; that's safe because the entry was read whole.
322
+ * `settle` recomputes the touched URLs right away (hooks); a full scan settles them at the end.
323
+ */
324
+ async function refreshEntry(ctx, plans, plan, item, scanId, settle) {
325
+ if (item.status !== "published") return retireEntry(ctx, plans, plan.slug, item.id);
326
+ const now = (/* @__PURE__ */ new Date()).toISOString();
327
+ const nodes = nodesOf(ctx);
328
+ const edges = edgesOf(ctx);
329
+ const contentId = contentNodeId(plan.slug, item.id);
330
+ const isDefaultLocale = !item.locale || item.locale === ctx.site.locale;
331
+ const path = plan.routable && item.slug && isDefaultLocale ? entryPath(plan.pattern, plan.slug, item.slug, item.id) : null;
332
+ const links = extractLinks(item.data, plan.urlFields).flatMap((link) => {
333
+ const target = internalPath(link.href, path ?? "/", ctx.site.url);
334
+ return target && target !== path ? [{
335
+ ...link,
336
+ target
337
+ }] : [];
338
+ });
339
+ const nodeIds = [
340
+ contentId,
341
+ ...path ? [urlNodeId(path)] : [],
342
+ ...links.map((l) => urlNodeId(l.target))
343
+ ];
344
+ const existing = await nodes.getMany([...new Set(nodeIds)]);
345
+ const nodeDocs = /* @__PURE__ */ new Map();
346
+ nodeDocs.set(contentId, discoveredNode(existing.get(contentId), {
347
+ type: "CONTENT",
348
+ label: entryLabel(item, plan),
349
+ ref: `${plan.slug}/${item.id}`
350
+ }, now, scanId));
351
+ if (path) nodeDocs.set(urlNodeId(path), discoveredNode(existing.get(urlNodeId(path)), {
352
+ type: "URL",
353
+ label: path,
354
+ ref: path,
355
+ resolved: true
356
+ }, now, scanId));
357
+ for (const link of links) {
358
+ const id = urlNodeId(link.target);
359
+ if (nodeDocs.has(id)) continue;
360
+ const prior = existing.get(id);
361
+ const resolved = prior?.active ? prior.resolved : looksLikeEntryPath(link.target, plans) ? false : void 0;
362
+ const doc = discoveredNode(prior, {
363
+ type: "URL",
364
+ label: link.target,
365
+ ref: link.target
366
+ }, now, scanId);
367
+ if (resolved === void 0) delete doc.resolved;
368
+ else doc.resolved = resolved;
369
+ nodeDocs.set(id, doc);
370
+ }
371
+ const edgeDocs = /* @__PURE__ */ new Map();
372
+ const edge = (source, relation, target, field, evidence) => {
373
+ edgeDocs.set(edgeId(source, relation, target, field), {
374
+ sourceNodeId: source,
375
+ targetNodeId: target,
376
+ relation,
377
+ provenance: "DISCOVERED",
378
+ evidence: {
379
+ sourceId: `${plan.slug}/${item.id}`,
380
+ observedAt: now,
381
+ ...evidence
382
+ },
383
+ active: true,
384
+ scanId,
385
+ firstSeenAt: now,
386
+ lastSeenAt: now,
387
+ schemaVersion: SCHEMA_VERSION
388
+ });
389
+ };
390
+ if (path) edge(contentId, "PUBLISHES_AS", urlNodeId(path), "", { href: path });
391
+ for (const link of links) edge(contentId, "LINKS_TO", urlNodeId(link.target), link.field, {
392
+ fieldPath: link.path,
393
+ href: link.href
394
+ });
395
+ const previous = await queryAll(edges, {
396
+ sourceNodeId: contentId,
397
+ provenance: "DISCOVERED"
398
+ });
399
+ const prevById = new Map(previous.map((p) => [p.id, p.data]));
400
+ for (const [id, doc] of edgeDocs) {
401
+ const first = prevById.get(id)?.firstSeenAt;
402
+ if (first) doc.firstSeenAt = first;
403
+ }
404
+ const stale = previous.filter((p) => p.data.active && !edgeDocs.has(p.id));
405
+ await nodes.putMany([...nodeDocs].map(([id, data]) => ({
406
+ id,
407
+ data
408
+ })));
409
+ await edges.putMany([...[...edgeDocs].map(([id, data]) => ({
410
+ id,
411
+ data
412
+ })), ...stale.map((p) => ({
413
+ id: p.id,
414
+ data: {
415
+ ...p.data,
416
+ active: false
417
+ }
418
+ }))]);
419
+ if (settle) {
420
+ const touched = new Set([...stale.map((e) => e.data.targetNodeId), ...links.map((l) => urlNodeId(l.target))]);
421
+ for (const id of touched) await settleUrl(ctx, plans, id);
422
+ }
423
+ }
424
+ /** An entry left the published site: hide its node and links, then re-settle the URLs it touched. */
425
+ async function retireEntry(ctx, plans, collection, entryId) {
426
+ const nodes = nodesOf(ctx);
427
+ const edges = edgesOf(ctx);
428
+ const contentId = contentNodeId(collection, entryId);
429
+ const outgoing = (await queryAll(edges, {
430
+ sourceNodeId: contentId,
431
+ provenance: "DISCOVERED"
432
+ })).filter((e) => e.data.active);
433
+ await edges.putMany(outgoing.map((e) => ({
434
+ id: e.id,
435
+ data: {
436
+ ...e.data,
437
+ active: false
438
+ }
439
+ })));
440
+ const content = await nodes.get(contentId);
441
+ if (content?.active) await nodes.put(contentId, {
442
+ ...content,
443
+ active: false
444
+ });
445
+ for (const id of new Set(outgoing.map((e) => e.data.targetNodeId))) await settleUrl(ctx, plans, id);
446
+ }
447
+ /** Refresh one entry from a content hook. Never scans the site (SPEC D8). */
448
+ async function refreshFromHook(ctx, collection, entryId) {
449
+ const plans = await collectionPlans(ctx);
450
+ const plan = plans.find((p) => p.slug === collection);
451
+ if (!plan) return;
452
+ const item = await ctx.content?.get(collection, entryId);
453
+ if (!item) return retireEntry(ctx, plans, collection, entryId);
454
+ await refreshEntry(ctx, plans, plan, item, (await ctx.kv.get(SCAN_KEY))?.id ?? "hook", true);
455
+ }
456
+ async function retireFromHook(ctx, collection, entryId) {
457
+ await retireEntry(ctx, await collectionPlans(ctx), collection, entryId);
458
+ }
459
+ async function getScanStatus(ctx) {
460
+ return {
461
+ running: await ctx.kv.get(SCAN_KEY),
462
+ last: await ctx.kv.get(LAST_SCAN_KEY)
463
+ };
464
+ }
465
+ /** Start a scan, or join the one in progress (two admins clicking at once share it). */
466
+ async function startScan(ctx) {
467
+ const running = await ctx.kv.get(SCAN_KEY);
468
+ if (running && Date.now() - Date.parse(running.updatedAt ?? running.startedAt) < STALE_AFTER_MS) return running;
469
+ const now = (/* @__PURE__ */ new Date()).toISOString();
470
+ const state = {
471
+ id: `scan_${crypto.randomUUID()}`,
472
+ startedAt: now,
473
+ updatedAt: now,
474
+ phase: "collect",
475
+ collections: await collectionPlans(ctx),
476
+ index: 0,
477
+ processed: 0,
478
+ retired: 0,
479
+ errors: []
480
+ };
481
+ await ctx.kv.set(SCAN_KEY, state);
482
+ return state;
483
+ }
484
+ /**
485
+ * Advance scan `scanId` by one bounded batch. The admin calls this until `done`, so no single
486
+ * request runs long. Reconciliation runs only after every collection was read cleanly, so a
487
+ * failed or abandoned scan never retires anything (SPEC D7). Writes are compare-and-set: if
488
+ * another tab advanced the scan meanwhile, this step's state is dropped instead of rolling
489
+ * theirs back (the work it did is idempotent).
490
+ */
491
+ async function scanStep(ctx, scanId) {
492
+ const versioned = await ctx.kv.getVersioned(SCAN_KEY);
493
+ const state = versioned?.value;
494
+ if (!state) return {
495
+ state: null,
496
+ last: await ctx.kv.get(LAST_SCAN_KEY),
497
+ done: true
498
+ };
499
+ if (state.id !== scanId) throw PluginRouteError.conflict("A newer scan replaced this one. Reload to follow it.");
500
+ const plans = state.collections;
501
+ if (state.phase === "collect") {
502
+ const plan = plans[state.index];
503
+ if (!plan) {
504
+ state.phase = "reconcile-edges";
505
+ state.cursor = void 0;
506
+ } else try {
507
+ const page = await ctx.content.list(plan.slug, {
508
+ limit: BATCH,
509
+ cursor: state.cursor,
510
+ where: { status: "published" }
511
+ });
512
+ for (const item of page.items) try {
513
+ await refreshEntry(ctx, plans, plan, item, state.id, false);
514
+ state.processed++;
515
+ } catch (error) {
516
+ if (state.errors.length < MAX_ERRORS) state.errors.push(`${plan.slug}/${item.id}: ${errorText(error)}`);
517
+ }
518
+ if (page.hasMore && page.cursor) state.cursor = page.cursor;
519
+ else {
520
+ state.index++;
521
+ state.cursor = void 0;
522
+ }
523
+ } catch (error) {
524
+ if (state.errors.length < MAX_ERRORS) state.errors.push(`${plan.slug}: ${errorText(error)}`);
525
+ state.index++;
526
+ state.cursor = void 0;
527
+ }
528
+ } else if (state.errors.length) return finish(ctx, state);
529
+ else if (state.phase === "reconcile-edges") {
530
+ const page = await edgesOf(ctx).query({
531
+ where: { provenance: "DISCOVERED" },
532
+ limit: 100,
533
+ cursor: state.cursor
534
+ });
535
+ const stale = page.items.filter((row) => row.data.active && row.data.scanId !== state.id);
536
+ if (stale.length) {
537
+ await edgesOf(ctx).putMany(stale.map((row) => ({
538
+ id: row.id,
539
+ data: {
540
+ ...row.data,
541
+ active: false
542
+ }
543
+ })));
544
+ state.retired += stale.length;
545
+ }
546
+ if (page.hasMore && page.cursor) state.cursor = page.cursor;
547
+ else {
548
+ state.phase = "reconcile-nodes";
549
+ state.cursor = void 0;
550
+ }
551
+ } else {
552
+ const page = await nodesOf(ctx).query({
553
+ where: { provenance: "DISCOVERED" },
554
+ limit: 50,
555
+ cursor: state.cursor
556
+ });
557
+ for (const row of page.items) {
558
+ if (!row.data.active) continue;
559
+ if (row.data.type === "URL") {
560
+ if (await settleUrl(ctx, plans, row.id) && !(await nodesOf(ctx).get(row.id))?.active) state.retired++;
561
+ } else if (row.data.scanId !== state.id) {
562
+ await nodesOf(ctx).put(row.id, {
563
+ ...row.data,
564
+ active: false
565
+ });
566
+ state.retired++;
567
+ }
568
+ }
569
+ if (page.hasMore && page.cursor) state.cursor = page.cursor;
570
+ else return finish(ctx, state);
571
+ }
572
+ state.updatedAt = (/* @__PURE__ */ new Date()).toISOString();
573
+ const current = (await ctx.kv.compareAndSet(SCAN_KEY, versioned.revision, state)).applied ? state : await ctx.kv.get(SCAN_KEY);
574
+ return {
575
+ state: current,
576
+ last: await ctx.kv.get(LAST_SCAN_KEY),
577
+ done: !current
578
+ };
579
+ }
580
+ async function finish(ctx, state) {
581
+ const last = {
582
+ id: state.id,
583
+ startedAt: state.startedAt,
584
+ finishedAt: (/* @__PURE__ */ new Date()).toISOString(),
585
+ status: state.errors.length ? "PARTIAL" : "SUCCEEDED",
586
+ processed: state.processed,
587
+ retired: state.retired,
588
+ errors: state.errors
589
+ };
590
+ await ctx.kv.set(LAST_SCAN_KEY, last);
591
+ await ctx.kv.delete(SCAN_KEY);
592
+ return {
593
+ state: null,
594
+ last,
595
+ done: true
596
+ };
597
+ }
598
+
599
+ //#endregion
600
+ //#region src/routes.ts
601
+ const READ = "plugins:read";
602
+ const MANAGE = "plugins:manage";
603
+ const id = z.string().min(1).max(600);
604
+ const text = (max) => z.string().trim().max(max);
605
+ const safeUrl = z.string().trim().max(2e3).refine((v) => v === "" || /^https?:\/\//i.test(v), "Only http(s) links are allowed");
606
+ const annotation = {
607
+ description: text(2e3).optional(),
608
+ criticality: z.enum(CRITICALITIES).nullable().optional(),
609
+ notes: text(5e3).optional(),
610
+ docUrl: safeUrl.optional(),
611
+ lastVerifiedAt: z.iso.datetime().nullable().optional()
612
+ };
613
+ const nodeSaveInput = z.object({
614
+ id: id.optional(),
615
+ type: z.enum(DOCUMENTED_NODE_TYPES).optional(),
616
+ label: text(200).min(1).optional(),
617
+ ...annotation
618
+ });
619
+ const NEIGHBOUR_CAP = 200;
620
+ async function nodesById(ctx, ids) {
621
+ return [...await nodesOf(ctx).getMany([...new Set(ids)])].map(([nodeId, data]) => ({
622
+ id: nodeId,
623
+ ...data
624
+ }));
625
+ }
626
+ async function requireNode(ctx, nodeId) {
627
+ const node = await nodesOf(ctx).get(nodeId);
628
+ if (!node || !node.active) throw PluginRouteError.notFound("That node doesn't exist (or a scan retired it).");
629
+ return node;
630
+ }
631
+ function applyAnnotation(target, input) {
632
+ const out = { ...target };
633
+ for (const key of Object.keys(annotation)) {
634
+ const value = input[key];
635
+ if (value === void 0) continue;
636
+ if (value === "" || value === null) delete out[key];
637
+ else out[key] = value;
638
+ }
639
+ return out;
640
+ }
641
+ const routes = {
642
+ health: {
643
+ permission: READ,
644
+ handler: async (ctx) => ({
645
+ plugin: ctx.plugin.id,
646
+ version: ctx.plugin.version,
647
+ siteUrl: ctx.site.url || null
648
+ })
649
+ },
650
+ overview: {
651
+ permission: READ,
652
+ handler: async (ctx) => {
653
+ const nodes = nodesOf(ctx);
654
+ const counts = Object.fromEntries(await Promise.all(NODE_TYPES.map(async (type) => [type, await nodes.count({
655
+ type,
656
+ active: true
657
+ })])));
658
+ return {
659
+ siteUrl: ctx.site.url || null,
660
+ counts,
661
+ edges: await edgesOf(ctx).count({ active: true }),
662
+ brokenLinks: await nodes.count({
663
+ type: "URL",
664
+ active: true,
665
+ resolved: false
666
+ }),
667
+ scan: await getScanStatus(ctx)
668
+ };
669
+ }
670
+ },
671
+ "scan/start": {
672
+ permission: MANAGE,
673
+ handler: async (ctx) => ({ state: await startScan(ctx) })
674
+ },
675
+ "scan/step": {
676
+ permission: MANAGE,
677
+ input: z.object({ scanId: z.string().min(1).max(100) }),
678
+ handler: async (ctx) => scanStep(ctx, ctx.input.scanId)
679
+ },
680
+ "nodes/search": {
681
+ permission: READ,
682
+ input: z.object({
683
+ q: text(200).default(""),
684
+ type: z.enum(NODE_TYPES).optional(),
685
+ broken: z.boolean().optional(),
686
+ cursor: z.string().optional()
687
+ }),
688
+ handler: async (ctx) => {
689
+ const { q, type, broken, cursor } = ctx.input;
690
+ const where = { active: true };
691
+ if (q) where.search = { startsWith: q.toLowerCase() };
692
+ if (type) where.type = type;
693
+ if (broken) Object.assign(where, {
694
+ type: "URL",
695
+ resolved: false
696
+ });
697
+ const page = await nodesOf(ctx).query({
698
+ where,
699
+ limit: 50,
700
+ cursor
701
+ });
702
+ return {
703
+ items: page.items.map((row) => ({
704
+ id: row.id,
705
+ ...row.data
706
+ })),
707
+ cursor: page.hasMore ? page.cursor : null
708
+ };
709
+ }
710
+ },
711
+ "graph/neighborhood": {
712
+ permission: READ,
713
+ input: z.object({ nodeId: id }),
714
+ handler: async (ctx) => {
715
+ const center = await requireNode(ctx, ctx.input.nodeId);
716
+ const cap = NEIGHBOUR_CAP + 1;
717
+ const [outbound, inbound] = await Promise.all([activeEdges(ctx, [ctx.input.nodeId], "outbound", cap), activeEdges(ctx, [ctx.input.nodeId], "inbound", cap)]);
718
+ const ownUrls = outbound.filter((e) => e.relation === "PUBLISHES_AS").map((e) => e.targetNodeId);
719
+ const linkers = ownUrls.length ? await activeEdges(ctx, ownUrls, "inbound", cap) : [];
720
+ const all = [...new Map([
721
+ ...outbound,
722
+ ...inbound,
723
+ ...linkers
724
+ ].map((e) => [e.id, e])).values()];
725
+ const edges = all.slice(0, NEIGHBOUR_CAP);
726
+ const nodes = await nodesById(ctx, edges.flatMap((e) => [e.sourceNodeId, e.targetNodeId]));
727
+ return {
728
+ center: {
729
+ id: ctx.input.nodeId,
730
+ ...center
731
+ },
732
+ nodes,
733
+ edges,
734
+ truncated: all.length > edges.length,
735
+ total: all.length
736
+ };
737
+ }
738
+ },
739
+ "graph/impact": {
740
+ permission: READ,
741
+ input: z.object({
742
+ nodeId: id,
743
+ direction: z.enum([
744
+ "inbound",
745
+ "outbound",
746
+ "both"
747
+ ]).default("inbound"),
748
+ depth: z.number().int().min(1).max(MAX_DEPTH$1).default(2)
749
+ }),
750
+ handler: async (ctx) => {
751
+ await requireNode(ctx, ctx.input.nodeId);
752
+ const result = await impact(ctx.input.nodeId, ctx.input.direction, ctx.input.depth, (ids, dir) => activeEdges(ctx, ids, dir, MAX_EDGES));
753
+ const nodes = await nodesById(ctx, [result.start, ...result.hits.map((h) => h.nodeId)]);
754
+ return {
755
+ ...result,
756
+ nodes
757
+ };
758
+ }
759
+ },
760
+ "nodes/save": {
761
+ permission: MANAGE,
762
+ input: nodeSaveInput,
763
+ handler: async (ctx) => {
764
+ const nodes = nodesOf(ctx);
765
+ const now = (/* @__PURE__ */ new Date()).toISOString();
766
+ const { id: nodeId, type, label } = ctx.input;
767
+ if (nodeId) {
768
+ const existing = await requireNode(ctx, nodeId);
769
+ const renamed = existing.provenance === "DOCUMENTED" && label ? {
770
+ label,
771
+ search: label.toLowerCase()
772
+ } : {};
773
+ const updated = applyAnnotation({
774
+ ...existing,
775
+ ...renamed
776
+ }, ctx.input);
777
+ await nodes.put(nodeId, updated);
778
+ return {
779
+ id: nodeId,
780
+ ...updated
781
+ };
782
+ }
783
+ if (!type || !label) throw PluginRouteError.badRequest("A new node needs a type and a label.");
784
+ const created = applyAnnotation({
785
+ type,
786
+ label,
787
+ search: label.toLowerCase(),
788
+ ref: label,
789
+ provenance: "DOCUMENTED",
790
+ active: true,
791
+ firstSeenAt: now,
792
+ lastSeenAt: now,
793
+ schemaVersion: SCHEMA_VERSION
794
+ }, ctx.input);
795
+ const newId = `doc:${crypto.randomUUID()}`;
796
+ await nodes.put(newId, created);
797
+ return {
798
+ id: newId,
799
+ ...created
800
+ };
801
+ }
802
+ },
803
+ "nodes/delete": {
804
+ permission: MANAGE,
805
+ input: z.object({ id }),
806
+ handler: async (ctx) => {
807
+ if ((await requireNode(ctx, ctx.input.id)).provenance !== "DOCUMENTED") throw PluginRouteError.badRequest("Discovered nodes come from your content. Change the content instead.");
808
+ const touching = [...await queryAll(edgesOf(ctx), { sourceNodeId: ctx.input.id }), ...await queryAll(edgesOf(ctx), { targetNodeId: ctx.input.id })];
809
+ await edgesOf(ctx).deleteMany(touching.map((e) => e.id));
810
+ await nodesOf(ctx).delete(ctx.input.id);
811
+ return {
812
+ deleted: ctx.input.id,
813
+ edgesDeleted: touching.length
814
+ };
815
+ }
816
+ },
817
+ "edges/save": {
818
+ permission: MANAGE,
819
+ input: z.object({
820
+ sourceNodeId: id,
821
+ targetNodeId: id,
822
+ relation: z.enum(DOCUMENTED_RELATION_TYPES),
823
+ label: text(200).optional()
824
+ }),
825
+ handler: async (ctx) => {
826
+ const { sourceNodeId, targetNodeId, relation, label } = ctx.input;
827
+ if (sourceNodeId === targetNodeId) throw PluginRouteError.badRequest("A node can't depend on itself.");
828
+ if (relation === "RELATED_TO" && !label) throw PluginRouteError.badRequest("Say how they're related: RELATED_TO needs a label.");
829
+ await requireNode(ctx, sourceNodeId);
830
+ await requireNode(ctx, targetNodeId);
831
+ const now = (/* @__PURE__ */ new Date()).toISOString();
832
+ const edgeKey = edgeId(sourceNodeId, relation, targetNodeId, "doc");
833
+ const existing = await edgesOf(ctx).get(edgeKey);
834
+ const edge = {
835
+ sourceNodeId,
836
+ targetNodeId,
837
+ relation,
838
+ ...label ? { label } : {},
839
+ provenance: "DOCUMENTED",
840
+ evidence: {
841
+ observedAt: now,
842
+ ...ctx.user?.id ? { sourceId: `user:${ctx.user.id}` } : {}
843
+ },
844
+ active: true,
845
+ firstSeenAt: existing?.firstSeenAt ?? now,
846
+ lastSeenAt: now,
847
+ schemaVersion: SCHEMA_VERSION
848
+ };
849
+ await edgesOf(ctx).put(edgeKey, edge);
850
+ return {
851
+ id: edgeKey,
852
+ ...edge
853
+ };
854
+ }
855
+ },
856
+ "edges/delete": {
857
+ permission: MANAGE,
858
+ input: z.object({ id }),
859
+ handler: async (ctx) => {
860
+ const edge = await edgesOf(ctx).get(ctx.input.id);
861
+ if (!edge) throw PluginRouteError.notFound("That relationship doesn't exist.");
862
+ if (edge.provenance !== "DOCUMENTED") throw PluginRouteError.badRequest("Discovered links come from your content. Edit the content instead.");
863
+ await edgesOf(ctx).delete(ctx.input.id);
864
+ return { deleted: ctx.input.id };
865
+ }
866
+ },
867
+ export: {
868
+ permission: READ,
869
+ handler: async (ctx) => ({
870
+ format: "sitegraph",
871
+ schemaVersion: SCHEMA_VERSION,
872
+ plugin: {
873
+ id: ctx.plugin.id,
874
+ version: ctx.plugin.version
875
+ },
876
+ site: ctx.site.url || null,
877
+ exportedAt: (/* @__PURE__ */ new Date()).toISOString(),
878
+ nodes: (await queryAll(nodesOf(ctx), { active: true })).map((r) => ({
879
+ id: r.id,
880
+ ...r.data
881
+ })),
882
+ edges: (await queryAll(edgesOf(ctx), { active: true })).map((r) => ({
883
+ id: r.id,
884
+ ...r.data
885
+ }))
886
+ })
887
+ }
888
+ };
889
+
890
+ //#endregion
891
+ //#region src/index.ts
892
+ const ID = "sitegraph";
893
+ const VERSION = "0.1.0";
894
+ const PACKAGE = "emdash-plugin-sitegraph";
895
+ const ADMIN_ENTRY = `${PACKAGE}/admin`;
896
+ const PAGES = [{
897
+ path: "/",
898
+ label: "SiteGraph",
899
+ icon: "graph"
900
+ }];
901
+ const WIDGETS = [{
902
+ id: "overview",
903
+ title: "SiteGraph",
904
+ size: "half"
905
+ }];
906
+ /** Build-time descriptor: register with `emdash({ plugins: [siteGraph()] })`. */
907
+ function siteGraph() {
908
+ return {
909
+ id: ID,
910
+ version: VERSION,
911
+ format: "native",
912
+ entrypoint: PACKAGE,
913
+ adminEntry: ADMIN_ENTRY,
914
+ adminPages: PAGES,
915
+ adminWidgets: WIDGETS
916
+ };
917
+ }
918
+ const entryId = (content) => typeof content.id === "string" ? content.id : null;
919
+ /** Runtime entry. EmDash imports this by name. */
920
+ function createPlugin() {
921
+ return definePlugin({
922
+ id: ID,
923
+ version: VERSION,
924
+ capabilities: ["content:read", "schema:read"],
925
+ storage: STORAGE,
926
+ admin: {
927
+ entry: ADMIN_ENTRY,
928
+ pages: PAGES,
929
+ widgets: WIDGETS
930
+ },
931
+ hooks: {
932
+ "content:afterSave": async (event, ctx) => {
933
+ const id = entryId(event.content);
934
+ if (id) await refreshFromHook(ctx, event.collection, id);
935
+ },
936
+ "content:afterPublish": async (event, ctx) => {
937
+ const id = entryId(event.content);
938
+ if (id) await refreshFromHook(ctx, event.collection, id);
939
+ },
940
+ "content:afterRestore": async (event, ctx) => {
941
+ const id = entryId(event.content);
942
+ if (id) await refreshFromHook(ctx, event.collection, id);
943
+ },
944
+ "content:afterUnpublish": async (event, ctx) => {
945
+ const id = entryId(event.content);
946
+ if (id) await retireFromHook(ctx, event.collection, id);
947
+ },
948
+ "content:afterDelete": async (event, ctx) => {
949
+ await retireFromHook(ctx, event.collection, event.id);
950
+ }
951
+ },
952
+ routes
953
+ });
954
+ }
955
+
956
+ //#endregion
957
+ export { createPlugin, siteGraph as default, siteGraph };