emdash-plugin-sitegraph 0.0.0-stage → 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/src/scan.ts ADDED
@@ -0,0 +1,379 @@
1
+ // Discovery: turns published EmDash entries into CONTENT/URL nodes and PUBLISHES_AS/LINKS_TO edges.
2
+ // Read-only towards content (SPEC D3). Rescans are idempotent thanks to deterministic IDs.
3
+
4
+ import type { CollectionSchemaInfo, PluginContext } from "emdash";
5
+ import { PluginRouteError } from "emdash";
6
+
7
+ import {
8
+ contentNodeId,
9
+ edgeId,
10
+ type GraphEdge,
11
+ type GraphNode,
12
+ SCHEMA_VERSION,
13
+ urlNodeId,
14
+ } from "./domain/graph.js";
15
+ import { entryPath, entryPathMatcher, extractLinks, internalPath } from "./domain/links.js";
16
+ import { activeEdges, edgesOf, nodesOf, queryAll } from "./store.js";
17
+
18
+ /** The slice of a plugin-visible content item that discovery reads. */
19
+ interface Entry {
20
+ id: string;
21
+ slug: string | null;
22
+ status: string;
23
+ locale?: string | null;
24
+ data: Record<string, unknown>;
25
+ }
26
+
27
+ export interface CollectionPlan {
28
+ slug: string;
29
+ label: string;
30
+ pattern: string | null;
31
+ routable: boolean;
32
+ titleField: string | null;
33
+ urlFields: string[];
34
+ }
35
+
36
+ export interface ScanState {
37
+ id: string;
38
+ startedAt: string;
39
+ /** Heartbeat: a scan nobody has advanced for a while can be replaced. */
40
+ updatedAt: string;
41
+ phase: "collect" | "reconcile-edges" | "reconcile-nodes";
42
+ collections: CollectionPlan[];
43
+ index: number;
44
+ cursor?: string;
45
+ processed: number;
46
+ retired: number;
47
+ errors: string[];
48
+ }
49
+
50
+ export interface LastScan {
51
+ id: string;
52
+ startedAt: string;
53
+ finishedAt: string;
54
+ status: "SUCCEEDED" | "PARTIAL";
55
+ processed: number;
56
+ retired: number;
57
+ errors: string[];
58
+ }
59
+
60
+ const SCAN_KEY = "scan";
61
+ const LAST_SCAN_KEY = "lastScan";
62
+ const BATCH = 25;
63
+ const STALE_AFTER_MS = 2 * 60 * 1000;
64
+ const MAX_ERRORS = 20;
65
+
66
+ const errorText = (error: unknown) => (error instanceof Error ? error.message : String(error));
67
+
68
+ export async function collectionPlans(ctx: PluginContext): Promise<CollectionPlan[]> {
69
+ const collections: CollectionSchemaInfo[] = (await ctx.schema?.listCollections()) ?? [];
70
+ return collections.map((c) => ({
71
+ slug: c.slug,
72
+ label: c.labelSingular ?? c.label,
73
+ pattern: c.urlPattern,
74
+ routable: c.routable,
75
+ titleField: c.titleField,
76
+ urlFields: c.fields.filter((f) => f.type === "url").map((f) => f.slug),
77
+ }));
78
+ }
79
+
80
+ /**
81
+ * Could an entry live at this path? Yes when it fits a routable collection's URL pattern and
82
+ * doesn't look like a file. A link to such a path with no entry behind it is broken; a link
83
+ * to anything else (home page, listings, feeds) points at a page we simply don't map.
84
+ */
85
+ export function looksLikeEntryPath(path: string, plans: CollectionPlan[]): boolean {
86
+ const last = path.slice(path.lastIndexOf("/") + 1);
87
+ if (last.includes(".")) return false;
88
+ return plans.some((p) => p.routable && entryPathMatcher(p.pattern, p.slug)?.test(path));
89
+ }
90
+
91
+ function entryLabel(item: Entry, plan: CollectionPlan): string {
92
+ const title = plan.titleField ? item.data[plan.titleField] : item.data.title;
93
+ if (typeof title === "string" && title.trim()) return title.trim().slice(0, 200);
94
+ return item.slug ?? item.id;
95
+ }
96
+
97
+ /** A discovered node merged over what's stored, so human annotations and firstSeenAt survive. */
98
+ function discoveredNode(
99
+ existing: GraphNode | undefined,
100
+ fresh: Pick<GraphNode, "type" | "label" | "ref"> & Partial<GraphNode>,
101
+ now: string,
102
+ scanId: string,
103
+ ): GraphNode & { scanId: string } {
104
+ return {
105
+ ...existing,
106
+ ...fresh,
107
+ search: fresh.label.toLowerCase(),
108
+ provenance: "DISCOVERED",
109
+ active: true,
110
+ firstSeenAt: existing?.firstSeenAt ?? now,
111
+ lastSeenAt: now,
112
+ schemaVersion: SCHEMA_VERSION,
113
+ scanId,
114
+ };
115
+ }
116
+
117
+ /**
118
+ * Recompute a URL node's status from the edges that point at it, the single source of truth:
119
+ * a published entry there → resolved; otherwise broken if it looks like an entry path, or
120
+ * unknown if not; and with nothing pointing at it at all, it leaves the map.
121
+ */
122
+ export async function settleUrl(ctx: PluginContext, plans: CollectionPlan[], nodeId: string): Promise<boolean> {
123
+ const nodes = nodesOf(ctx);
124
+ const node = await nodes.get(nodeId);
125
+ if (!node || node.type !== "URL" || node.provenance !== "DISCOVERED") return false;
126
+ const inbound = await activeEdges(ctx, [nodeId], "inbound", 50);
127
+ const hasPage = inbound.some((e) => e.relation === "PUBLISHES_AS");
128
+ const next = { ...node };
129
+ if (!hasPage && inbound.length === 0) next.active = false;
130
+ else if (hasPage) next.resolved = true;
131
+ else if (looksLikeEntryPath(node.ref, plans)) next.resolved = false;
132
+ else delete next.resolved;
133
+ if (next.active === node.active && next.resolved === node.resolved) return false;
134
+ await nodes.put(nodeId, next);
135
+ return true;
136
+ }
137
+
138
+ /**
139
+ * Re-read one entry's graph: its CONTENT node, its URL, and its outgoing links. Outgoing
140
+ * discovered edges it no longer has are retired; that's safe because the entry was read whole.
141
+ * `settle` recomputes the touched URLs right away (hooks); a full scan settles them at the end.
142
+ */
143
+ export async function refreshEntry(
144
+ ctx: PluginContext,
145
+ plans: CollectionPlan[],
146
+ plan: CollectionPlan,
147
+ item: Entry,
148
+ scanId: string,
149
+ settle: boolean,
150
+ ): Promise<void> {
151
+ if (item.status !== "published") return retireEntry(ctx, plans, plan.slug, item.id);
152
+
153
+ const now = new Date().toISOString();
154
+ const nodes = nodesOf(ctx);
155
+ const edges = edgesOf(ctx);
156
+ const contentId = contentNodeId(plan.slug, item.id);
157
+ // Translations are mapped as entries without a URL: we don't reproduce EmDash's locale
158
+ // prefixes, and guessing would collide translations onto one path (see README limits).
159
+ const isDefaultLocale = !item.locale || item.locale === ctx.site.locale;
160
+ const path =
161
+ plan.routable && item.slug && isDefaultLocale ? entryPath(plan.pattern, plan.slug, item.slug, item.id) : null;
162
+
163
+ const links = extractLinks(item.data, plan.urlFields).flatMap((link) => {
164
+ const target = internalPath(link.href, path ?? "/", ctx.site.url);
165
+ return target && target !== path ? [{ ...link, target }] : [];
166
+ });
167
+
168
+ const nodeIds = [contentId, ...(path ? [urlNodeId(path)] : []), ...links.map((l) => urlNodeId(l.target))];
169
+ const existing = await nodes.getMany([...new Set(nodeIds)]);
170
+
171
+ const nodeDocs = new Map<string, GraphNode>();
172
+ nodeDocs.set(
173
+ contentId,
174
+ discoveredNode(existing.get(contentId), { type: "CONTENT", label: entryLabel(item, plan), ref: `${plan.slug}/${item.id}` }, now, scanId),
175
+ );
176
+ if (path) {
177
+ nodeDocs.set(urlNodeId(path), discoveredNode(existing.get(urlNodeId(path)), { type: "URL", label: path, ref: path, resolved: true }, now, scanId));
178
+ }
179
+ for (const link of links) {
180
+ const id = urlNodeId(link.target);
181
+ if (nodeDocs.has(id)) continue;
182
+ const prior = existing.get(id);
183
+ // A first guess for brand-new URLs; settleUrl has the final word.
184
+ const resolved = prior?.active ? prior.resolved : looksLikeEntryPath(link.target, plans) ? false : undefined;
185
+ const doc = discoveredNode(prior, { type: "URL", label: link.target, ref: link.target }, now, scanId);
186
+ if (resolved === undefined) delete doc.resolved;
187
+ else doc.resolved = resolved;
188
+ nodeDocs.set(id, doc);
189
+ }
190
+
191
+ const edgeDocs = new Map<string, GraphEdge>();
192
+ const edge = (source: string, relation: GraphEdge["relation"], target: string, field: string, evidence: GraphEdge["evidence"]) => {
193
+ edgeDocs.set(edgeId(source, relation, target, field), {
194
+ sourceNodeId: source,
195
+ targetNodeId: target,
196
+ relation,
197
+ provenance: "DISCOVERED",
198
+ evidence: { sourceId: `${plan.slug}/${item.id}`, observedAt: now, ...evidence },
199
+ active: true,
200
+ scanId,
201
+ firstSeenAt: now,
202
+ lastSeenAt: now,
203
+ schemaVersion: SCHEMA_VERSION,
204
+ });
205
+ };
206
+ if (path) edge(contentId, "PUBLISHES_AS", urlNodeId(path), "", { href: path });
207
+ for (const link of links) edge(contentId, "LINKS_TO", urlNodeId(link.target), link.field, { fieldPath: link.path, href: link.href });
208
+
209
+ const previous = await queryAll(edges, { sourceNodeId: contentId, provenance: "DISCOVERED" });
210
+ const prevById = new Map(previous.map((p) => [p.id, p.data]));
211
+ for (const [id, doc] of edgeDocs) {
212
+ const first = prevById.get(id)?.firstSeenAt;
213
+ if (first) doc.firstSeenAt = first;
214
+ }
215
+ const stale = previous.filter((p) => p.data.active && !edgeDocs.has(p.id));
216
+
217
+ await nodes.putMany([...nodeDocs].map(([id, data]) => ({ id, data })));
218
+ await edges.putMany([
219
+ ...[...edgeDocs].map(([id, data]) => ({ id, data })),
220
+ ...stale.map((p) => ({ id: p.id, data: { ...p.data, active: false } })),
221
+ ]);
222
+
223
+ if (settle) {
224
+ // A renamed slug or a removed link changes the status of the URLs left behind.
225
+ const touched = new Set([...stale.map((e) => e.data.targetNodeId), ...links.map((l) => urlNodeId(l.target))]);
226
+ for (const id of touched) await settleUrl(ctx, plans, id);
227
+ }
228
+ }
229
+
230
+ /** An entry left the published site: hide its node and links, then re-settle the URLs it touched. */
231
+ export async function retireEntry(ctx: PluginContext, plans: CollectionPlan[], collection: string, entryId: string): Promise<void> {
232
+ const nodes = nodesOf(ctx);
233
+ const edges = edgesOf(ctx);
234
+ const contentId = contentNodeId(collection, entryId);
235
+ const outgoing = (await queryAll(edges, { sourceNodeId: contentId, provenance: "DISCOVERED" })).filter((e) => e.data.active);
236
+ await edges.putMany(outgoing.map((e) => ({ id: e.id, data: { ...e.data, active: false } })));
237
+ const content = await nodes.get(contentId);
238
+ if (content?.active) await nodes.put(contentId, { ...content, active: false });
239
+ for (const id of new Set(outgoing.map((e) => e.data.targetNodeId))) await settleUrl(ctx, plans, id);
240
+ }
241
+
242
+ /** Refresh one entry from a content hook. Never scans the site (SPEC D8). */
243
+ export async function refreshFromHook(ctx: PluginContext, collection: string, entryId: string): Promise<void> {
244
+ const plans = await collectionPlans(ctx);
245
+ const plan = plans.find((p) => p.slug === collection);
246
+ if (!plan) return;
247
+ const item = await ctx.content?.get(collection, entryId);
248
+ if (!item) return retireEntry(ctx, plans, collection, entryId);
249
+ const running = await ctx.kv.get<ScanState>(SCAN_KEY);
250
+ await refreshEntry(ctx, plans, plan, item, running?.id ?? "hook", true);
251
+ }
252
+
253
+ export async function retireFromHook(ctx: PluginContext, collection: string, entryId: string): Promise<void> {
254
+ await retireEntry(ctx, await collectionPlans(ctx), collection, entryId);
255
+ }
256
+
257
+ export async function getScanStatus(ctx: PluginContext): Promise<{ running: ScanState | null; last: LastScan | null }> {
258
+ return {
259
+ running: await ctx.kv.get<ScanState>(SCAN_KEY),
260
+ last: await ctx.kv.get<LastScan>(LAST_SCAN_KEY),
261
+ };
262
+ }
263
+
264
+ /** Start a scan, or join the one in progress (two admins clicking at once share it). */
265
+ export async function startScan(ctx: PluginContext): Promise<ScanState> {
266
+ const running = await ctx.kv.get<ScanState>(SCAN_KEY);
267
+ if (running && Date.now() - Date.parse(running.updatedAt ?? running.startedAt) < STALE_AFTER_MS) return running;
268
+ const now = new Date().toISOString();
269
+ const state: ScanState = {
270
+ id: `scan_${crypto.randomUUID()}`,
271
+ startedAt: now,
272
+ updatedAt: now,
273
+ phase: "collect",
274
+ collections: await collectionPlans(ctx),
275
+ index: 0,
276
+ processed: 0,
277
+ retired: 0,
278
+ errors: [],
279
+ };
280
+ await ctx.kv.set(SCAN_KEY, state);
281
+ return state;
282
+ }
283
+
284
+ /**
285
+ * Advance scan `scanId` by one bounded batch. The admin calls this until `done`, so no single
286
+ * request runs long. Reconciliation runs only after every collection was read cleanly, so a
287
+ * failed or abandoned scan never retires anything (SPEC D7). Writes are compare-and-set: if
288
+ * another tab advanced the scan meanwhile, this step's state is dropped instead of rolling
289
+ * theirs back (the work it did is idempotent).
290
+ */
291
+ export async function scanStep(
292
+ ctx: PluginContext,
293
+ scanId: string,
294
+ ): Promise<{ state: ScanState | null; last: LastScan | null; done: boolean }> {
295
+ const versioned = await ctx.kv.getVersioned<ScanState>(SCAN_KEY);
296
+ const state = versioned?.value;
297
+ if (!state) return { state: null, last: await ctx.kv.get<LastScan>(LAST_SCAN_KEY), done: true };
298
+ if (state.id !== scanId) throw PluginRouteError.conflict("A newer scan replaced this one. Reload to follow it.");
299
+ const plans = state.collections;
300
+
301
+ if (state.phase === "collect") {
302
+ const plan = plans[state.index];
303
+ if (!plan) {
304
+ state.phase = "reconcile-edges";
305
+ state.cursor = undefined;
306
+ } else {
307
+ try {
308
+ const page = await ctx.content!.list(plan.slug, { limit: BATCH, cursor: state.cursor, where: { status: "published" } });
309
+ for (const item of page.items) {
310
+ try {
311
+ await refreshEntry(ctx, plans, plan, item, state.id, false);
312
+ state.processed++;
313
+ } catch (error) {
314
+ if (state.errors.length < MAX_ERRORS) state.errors.push(`${plan.slug}/${item.id}: ${errorText(error)}`);
315
+ }
316
+ }
317
+ if (page.hasMore && page.cursor) state.cursor = page.cursor;
318
+ else {
319
+ state.index++;
320
+ state.cursor = undefined;
321
+ }
322
+ } catch (error) {
323
+ // The whole collection couldn't be listed (e.g. deleted mid-scan): note it and move on.
324
+ if (state.errors.length < MAX_ERRORS) state.errors.push(`${plan.slug}: ${errorText(error)}`);
325
+ state.index++;
326
+ state.cursor = undefined;
327
+ }
328
+ }
329
+ } else if (state.errors.length) {
330
+ // Failed entries' links weren't seen, so reconciling would wrongly retire them.
331
+ return finish(ctx, state);
332
+ } else if (state.phase === "reconcile-edges") {
333
+ const page = await edgesOf(ctx).query({ where: { provenance: "DISCOVERED" }, limit: 100, cursor: state.cursor });
334
+ const stale = page.items.filter((row) => row.data.active && row.data.scanId !== state.id);
335
+ if (stale.length) {
336
+ await edgesOf(ctx).putMany(stale.map((row) => ({ id: row.id, data: { ...row.data, active: false } })));
337
+ state.retired += stale.length;
338
+ }
339
+ if (page.hasMore && page.cursor) state.cursor = page.cursor;
340
+ else {
341
+ state.phase = "reconcile-nodes";
342
+ state.cursor = undefined;
343
+ }
344
+ } else {
345
+ // Entries this scan didn't see are gone; every URL is re-settled from its now-current edges.
346
+ const page = await nodesOf(ctx).query({ where: { provenance: "DISCOVERED" }, limit: 50, cursor: state.cursor });
347
+ for (const row of page.items) {
348
+ if (!row.data.active) continue;
349
+ if (row.data.type === "URL") {
350
+ if ((await settleUrl(ctx, plans, row.id)) && !(await nodesOf(ctx).get(row.id))?.active) state.retired++;
351
+ } else if ((row.data as { scanId?: string }).scanId !== state.id) {
352
+ await nodesOf(ctx).put(row.id, { ...row.data, active: false });
353
+ state.retired++;
354
+ }
355
+ }
356
+ if (page.hasMore && page.cursor) state.cursor = page.cursor;
357
+ else return finish(ctx, state);
358
+ }
359
+
360
+ state.updatedAt = new Date().toISOString();
361
+ const write = await ctx.kv.compareAndSet(SCAN_KEY, versioned!.revision, state);
362
+ const current = write.applied ? state : await ctx.kv.get<ScanState>(SCAN_KEY);
363
+ return { state: current, last: await ctx.kv.get<LastScan>(LAST_SCAN_KEY), done: !current };
364
+ }
365
+
366
+ async function finish(ctx: PluginContext, state: ScanState) {
367
+ const last: LastScan = {
368
+ id: state.id,
369
+ startedAt: state.startedAt,
370
+ finishedAt: new Date().toISOString(),
371
+ status: state.errors.length ? "PARTIAL" : "SUCCEEDED",
372
+ processed: state.processed,
373
+ retired: state.retired,
374
+ errors: state.errors,
375
+ };
376
+ await ctx.kv.set(LAST_SCAN_KEY, last);
377
+ await ctx.kv.delete(SCAN_KEY);
378
+ return { state: null, last, done: true };
379
+ }
package/src/store.ts ADDED
@@ -0,0 +1,55 @@
1
+ import type { PluginContext, StorageCollection } from "emdash";
2
+
3
+ import type { EdgeRecord } from "./domain/impact.js";
4
+ import type { GraphEdge, GraphNode } from "./domain/graph.js";
5
+
6
+ export const STORAGE = {
7
+ nodes: { indexes: ["type", "active", "search", "provenance", "resolved"] },
8
+ edges: { indexes: ["sourceNodeId", "targetNodeId", "provenance", "active"] },
9
+ };
10
+
11
+ export interface NodeRecord extends GraphNode {
12
+ id: string;
13
+ /** Scan that last saw a discovered node. */
14
+ scanId?: string;
15
+ }
16
+
17
+ export const nodesOf = (ctx: PluginContext) => ctx.storage.nodes as StorageCollection<Omit<NodeRecord, "id">>;
18
+ export const edgesOf = (ctx: PluginContext) => ctx.storage.edges as StorageCollection<GraphEdge>;
19
+
20
+ const PAGE = 100;
21
+
22
+ /** Documents matching `where`, page by page, stopping once `max` are read. */
23
+ export async function queryAll<T>(
24
+ collection: StorageCollection<T>,
25
+ where: Record<string, string | boolean | { in: string[] }>,
26
+ max = Infinity,
27
+ ): Promise<Array<{ id: string; data: T }>> {
28
+ const out: Array<{ id: string; data: T }> = [];
29
+ let cursor: string | undefined;
30
+ do {
31
+ const page = await collection.query({ where, limit: PAGE, cursor });
32
+ out.push(...page.items);
33
+ cursor = page.hasMore && out.length < max ? page.cursor : undefined;
34
+ } while (cursor);
35
+ return out.slice(0, max);
36
+ }
37
+
38
+ /**
39
+ * Active edges touching any of `nodeIds`, used by impact and neighbourhood views. Stops reading
40
+ * at `max`, so a hub page linked from every footer can't make a request read the whole table.
41
+ */
42
+ export async function activeEdges(
43
+ ctx: PluginContext,
44
+ nodeIds: string[],
45
+ direction: "inbound" | "outbound",
46
+ max = Infinity,
47
+ ): Promise<EdgeRecord[]> {
48
+ const field = direction === "outbound" ? "sourceNodeId" : "targetNodeId";
49
+ const out: EdgeRecord[] = [];
50
+ for (let i = 0; i < nodeIds.length && out.length < max; i += PAGE) {
51
+ const rows = await queryAll(edgesOf(ctx), { [field]: { in: nodeIds.slice(i, i + PAGE) }, active: true }, max - out.length);
52
+ for (const row of rows) out.push({ id: row.id, ...row.data });
53
+ }
54
+ return out;
55
+ }