@dmthepm/commune 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +468 -0
- package/bin/commune.mjs +29 -0
- package/lib/cli/check.d.ts +13 -0
- package/lib/cli/check.js +58 -0
- package/lib/cli/errors.d.ts +29 -0
- package/lib/cli/errors.js +41 -0
- package/lib/cli/gate.d.ts +34 -0
- package/lib/cli/gate.js +165 -0
- package/lib/cli/main.d.ts +20 -0
- package/lib/cli/main.js +175 -0
- package/lib/cli/query.d.ts +30 -0
- package/lib/cli/query.js +103 -0
- package/lib/cli/related.d.ts +17 -0
- package/lib/cli/related.js +177 -0
- package/lib/cli/render.d.ts +20 -0
- package/lib/cli/render.js +32 -0
- package/lib/cli/root.d.ts +11 -0
- package/lib/cli/root.js +28 -0
- package/lib/cli/usage.d.ts +3 -0
- package/lib/cli/usage.js +46 -0
- package/lib/cli/version.d.ts +24 -0
- package/lib/cli/version.js +29 -0
- package/lib/integration.d.ts +24 -0
- package/lib/integration.js +111 -0
- package/lib/lib/graph.d.ts +354 -0
- package/lib/lib/graph.js +774 -0
- package/lib/markdown.d.ts +30 -0
- package/lib/markdown.js +24 -0
- package/lib/rehype-external-links.d.ts +15 -0
- package/lib/rehype-external-links.js +46 -0
- package/lib/remark-wikilinks.d.ts +25 -0
- package/lib/remark-wikilinks.js +108 -0
- package/package.json +101 -0
- package/src/components/Backlinks.astro +17 -0
- package/src/components/BacklinksScript.astro +117 -0
- package/src/components/Footer.astro +35 -0
- package/src/components/Header.astro +250 -0
- package/src/components/HeaderStarScript.astro +380 -0
- package/src/components/HomeFooterCards.astro +155 -0
- package/src/components/MarkdownLink.astro +36 -0
- package/src/components/PlausibleScript.astro +13 -0
- package/src/components/RelatedNotes.astro +77 -0
- package/src/components/SearchModal.astro +213 -0
- package/src/components/StarredLinksScript.astro +92 -0
- package/src/components/panes.ts +61 -0
- package/src/styles/design-system.css +157 -0
- package/src/styles/notes.css +85 -0
package/lib/lib/graph.js
ADDED
|
@@ -0,0 +1,774 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The content graph core.
|
|
3
|
+
*
|
|
4
|
+
* Every mechanism that needs to know "what content exists and where does it
|
|
5
|
+
* live" reads from here: the remark WikiLinks plugin, the backlinks Astro
|
|
6
|
+
* integration, and the search-index check. Before this module those three
|
|
7
|
+
* carried hand-synced copies of the same directory scan, the same visibility
|
|
8
|
+
* rules, and three subtly different opinions about trailing slashes.
|
|
9
|
+
*
|
|
10
|
+
* The rules this module owns:
|
|
11
|
+
* - which collections participate (notes, research, pages)
|
|
12
|
+
* - visibility (notes opt in with `visibility: public`; research and pages
|
|
13
|
+
* are always public)
|
|
14
|
+
* - canonical URLs, always with a trailing slash, matching Astro's
|
|
15
|
+
* directory build format
|
|
16
|
+
* - the title/alias lookup used to resolve `[[WikiLinks]]`
|
|
17
|
+
* - which link forms count as an edge, and how each one resolves
|
|
18
|
+
*/
|
|
19
|
+
import { globby } from 'globby';
|
|
20
|
+
import { slug as githubSlug } from 'github-slugger';
|
|
21
|
+
import matter from 'gray-matter';
|
|
22
|
+
import { readFile } from 'node:fs/promises';
|
|
23
|
+
import path from 'node:path';
|
|
24
|
+
/** Where each collection's markdown lives, relative to the project root. */
|
|
25
|
+
export const CONTENT_DIRS = {
|
|
26
|
+
notes: 'src/content/notes',
|
|
27
|
+
research: 'src/content/research',
|
|
28
|
+
pages: 'src/content/pages',
|
|
29
|
+
};
|
|
30
|
+
/** Collection scan order. Stable so derived artifacts are deterministic. */
|
|
31
|
+
export const COLLECTIONS = ['notes', 'research', 'pages'];
|
|
32
|
+
/**
|
|
33
|
+
* Convert a content file path to its collection-relative slug and canonical URL.
|
|
34
|
+
*
|
|
35
|
+
* Files are named for their titles, so that `[[Evergreen Notes]]` resolves in
|
|
36
|
+
* Obsidian (which matches on filename) and in Astro (which matches on title)
|
|
37
|
+
* without a piped alias. The URL is therefore *derived* from the filename, not
|
|
38
|
+
* equal to it, using the same `github-slugger` call Astro's own content layer
|
|
39
|
+
* uses — so the graph and Astro's `entry.slug` can never disagree.
|
|
40
|
+
*
|
|
41
|
+
* Two escape hatches, in priority order:
|
|
42
|
+
* - `slug` frontmatter pins the URL when a rename would otherwise move it.
|
|
43
|
+
* Astro treats this key as reserved and strips it before Zod validation, so
|
|
44
|
+
* it must not appear in any collection schema.
|
|
45
|
+
* - `pages` declare a whole `url`, since they render at arbitrary routes.
|
|
46
|
+
*/
|
|
47
|
+
export function toUrlPath(file, collection, data = {}) {
|
|
48
|
+
const relative = path
|
|
49
|
+
.relative(CONTENT_DIRS[collection], file)
|
|
50
|
+
.replace(/\\/g, '/')
|
|
51
|
+
.replace(/\.(md|mdx)$/, '')
|
|
52
|
+
.replace(/\/index$/, '');
|
|
53
|
+
// Astro slugifies each path segment separately, so nested content keeps its
|
|
54
|
+
// directory structure. Match that exactly.
|
|
55
|
+
//
|
|
56
|
+
// The lambda is not ceremony: `githubSlug`'s second parameter is
|
|
57
|
+
// `maintainCase`, and passing the function straight to `map` would hand it
|
|
58
|
+
// the array index there — every segment after the first slugified with
|
|
59
|
+
// `maintainCase: 1`, which is truthy. Naming the one argument is what stops
|
|
60
|
+
// that.
|
|
61
|
+
const derived = relative
|
|
62
|
+
.split('/')
|
|
63
|
+
.map((segment) => githubSlug(segment))
|
|
64
|
+
.join('/');
|
|
65
|
+
const slug = typeof data.slug === 'string' && data.slug ? data.slug : derived;
|
|
66
|
+
if (collection === 'pages') {
|
|
67
|
+
return {
|
|
68
|
+
slug,
|
|
69
|
+
urlPath: typeof data.url === 'string' ? data.url : `/${slug}/`,
|
|
70
|
+
};
|
|
71
|
+
}
|
|
72
|
+
// Trailing slash on every route: Astro's directory build format emits
|
|
73
|
+
// `/notes/<slug>/index.html`, and the client-side graph lookups key off the
|
|
74
|
+
// pathname the browser actually reports.
|
|
75
|
+
return { slug, urlPath: `/${collection}/${slug}/` };
|
|
76
|
+
}
|
|
77
|
+
/**
|
|
78
|
+
* Convert a canonical URL path to the *file* path of its markdown twin.
|
|
79
|
+
*
|
|
80
|
+
* Every note is served twice at one URL: `/notes/cake/` renders the page,
|
|
81
|
+
* `/notes/cake.md` returns the source file. The mapping is pure string work —
|
|
82
|
+
* drop the trailing slash, append `.md` — and it lives here rather than in the
|
|
83
|
+
* build writer so the page templates cannot drift from the files that writer
|
|
84
|
+
* actually emits. The home page (`/`) has no name to hang the suffix on, so it
|
|
85
|
+
* gets `index.md`.
|
|
86
|
+
*
|
|
87
|
+
* The result is relative, with no leading slash: the caller joins it onto its
|
|
88
|
+
* output directory. For the URL a rendered page should link to, use
|
|
89
|
+
* `toMarkdownHref`. Traversal segments are rejected rather than normalized,
|
|
90
|
+
* since a `url` frontmatter is author-controlled and a `..` in it would
|
|
91
|
+
* otherwise let a build write outside its output directory.
|
|
92
|
+
*/
|
|
93
|
+
export function toMarkdownPath(urlPath) {
|
|
94
|
+
const segments = urlPath.split('/').filter((segment) => segment.length > 0);
|
|
95
|
+
if (segments.some((segment) => segment === '.' || segment === '..')) {
|
|
96
|
+
throw new Error(`Cannot map a URL path with a relative segment: ${urlPath}`);
|
|
97
|
+
}
|
|
98
|
+
return segments.length === 0 ? 'index.md' : `${segments.join('/')}.md`;
|
|
99
|
+
}
|
|
100
|
+
/**
|
|
101
|
+
* The site-absolute URL of an entry's markdown twin — what a rendered page links to.
|
|
102
|
+
*
|
|
103
|
+
* Derived from `toMarkdownPath` rather than computed alongside it, so a page's
|
|
104
|
+
* "view as markdown" link and the file the build writes can only ever be the
|
|
105
|
+
* same string with a leading slash.
|
|
106
|
+
*/
|
|
107
|
+
export function toMarkdownHref(urlPath) {
|
|
108
|
+
return `/${toMarkdownPath(urlPath)}`;
|
|
109
|
+
}
|
|
110
|
+
/** Coerce a frontmatter date to `yyyy-mm-dd`, so build output has no timestamp drift. */
|
|
111
|
+
export function normalizeDate(value) {
|
|
112
|
+
if (!value)
|
|
113
|
+
return undefined;
|
|
114
|
+
if (value instanceof Date)
|
|
115
|
+
return value.toISOString().split('T')[0];
|
|
116
|
+
if (typeof value === 'string')
|
|
117
|
+
return value.split('T')[0];
|
|
118
|
+
return undefined;
|
|
119
|
+
}
|
|
120
|
+
/** Is this entry publicly visible? Notes opt in; research and pages are always public. */
|
|
121
|
+
function isPublic(collection, data) {
|
|
122
|
+
if (collection !== 'notes')
|
|
123
|
+
return true;
|
|
124
|
+
return data.visibility === 'public';
|
|
125
|
+
}
|
|
126
|
+
/**
|
|
127
|
+
* Read every public content entry, in a stable order.
|
|
128
|
+
*
|
|
129
|
+
* Deliberately uncached: the backlinks integration runs this on each build hook
|
|
130
|
+
* and the dev server must see content edits without a restart.
|
|
131
|
+
*
|
|
132
|
+
* The root is a parameter rather than a `process.chdir`, so one process can
|
|
133
|
+
* read several vaults — which is what the CLI does when it is pointed at
|
|
134
|
+
* another wiki — without the cwd becoming shared mutable state.
|
|
135
|
+
*/
|
|
136
|
+
export async function loadContentEntries(options = {}) {
|
|
137
|
+
const root = options.root ?? process.cwd();
|
|
138
|
+
const entries = [];
|
|
139
|
+
for (const collection of COLLECTIONS) {
|
|
140
|
+
// `cwd` keeps globby's results root-relative, which is exactly the
|
|
141
|
+
// spelling `ContentEntry.file` promises; only the read needs the join.
|
|
142
|
+
const files = await globby(`${CONTENT_DIRS[collection]}/**/*.{md,mdx}`, { cwd: root });
|
|
143
|
+
files.sort();
|
|
144
|
+
for (const file of files) {
|
|
145
|
+
const source = await readFile(path.join(root, file), 'utf8');
|
|
146
|
+
const { content, data } = matter(source);
|
|
147
|
+
if (!isPublic(collection, data))
|
|
148
|
+
continue;
|
|
149
|
+
const { slug, urlPath } = toUrlPath(file, collection, data);
|
|
150
|
+
const summary = typeof data.summary === 'string' ? data.summary : undefined;
|
|
151
|
+
const updated = normalizeDate(data.updated);
|
|
152
|
+
entries.push({
|
|
153
|
+
slug,
|
|
154
|
+
urlPath,
|
|
155
|
+
title: data.title || slug,
|
|
156
|
+
collection,
|
|
157
|
+
aliases: data.aliases || [],
|
|
158
|
+
tags: data.tags || [],
|
|
159
|
+
status: data.status || 'seed',
|
|
160
|
+
...(summary ? { summary } : {}),
|
|
161
|
+
...(updated ? { updated } : {}),
|
|
162
|
+
body: content,
|
|
163
|
+
frontmatter: data,
|
|
164
|
+
file,
|
|
165
|
+
});
|
|
166
|
+
}
|
|
167
|
+
}
|
|
168
|
+
return entries;
|
|
169
|
+
}
|
|
170
|
+
/**
|
|
171
|
+
* Build the `[[WikiLink]]` resolution table: lowercased title or alias → target.
|
|
172
|
+
*
|
|
173
|
+
* Titles win over aliases, and within each pass later entries win — which only
|
|
174
|
+
* matters when two pieces of content claim the same name, and is a content bug
|
|
175
|
+
* either way.
|
|
176
|
+
*/
|
|
177
|
+
export function buildLinkLookup(entries) {
|
|
178
|
+
const lookup = new Map();
|
|
179
|
+
for (const entry of entries) {
|
|
180
|
+
const target = {
|
|
181
|
+
slug: entry.slug,
|
|
182
|
+
collection: entry.collection,
|
|
183
|
+
urlPath: entry.urlPath,
|
|
184
|
+
};
|
|
185
|
+
for (const alias of entry.aliases) {
|
|
186
|
+
lookup.set(alias.toLowerCase(), target);
|
|
187
|
+
}
|
|
188
|
+
lookup.set(entry.title.toLowerCase(), target);
|
|
189
|
+
}
|
|
190
|
+
return lookup;
|
|
191
|
+
}
|
|
192
|
+
/**
|
|
193
|
+
* Build the url resolution table: canonical `urlPath` → target.
|
|
194
|
+
*
|
|
195
|
+
* Separate from the title/alias lookup on purpose. A path and a title are
|
|
196
|
+
* different namespaces, and mixing them is what let `/notes/atomic-notes`
|
|
197
|
+
* resolve through an alias rather than through the route it actually names.
|
|
198
|
+
*/
|
|
199
|
+
export function buildUrlLookup(entries) {
|
|
200
|
+
const lookup = new Map();
|
|
201
|
+
for (const entry of entries) {
|
|
202
|
+
lookup.set(entry.urlPath, {
|
|
203
|
+
slug: entry.slug,
|
|
204
|
+
collection: entry.collection,
|
|
205
|
+
urlPath: entry.urlPath,
|
|
206
|
+
});
|
|
207
|
+
}
|
|
208
|
+
return lookup;
|
|
209
|
+
}
|
|
210
|
+
/**
|
|
211
|
+
* Build the filename resolution table: lowercased basename → target.
|
|
212
|
+
*
|
|
213
|
+
* Not a resolution table — nothing resolves through it. It exists so `check`
|
|
214
|
+
* can notice when a name means two different things depending on which table
|
|
215
|
+
* you ask: `[[Index]]` resolves by *title*, but `[index](./Index.md)` names a
|
|
216
|
+
* *file*, and in a vault where those disagree one of the two is silently wrong.
|
|
217
|
+
* Obsidian resolves by filename and Astro by title, so this is the seam where
|
|
218
|
+
* the two tools stop agreeing about what a link points at.
|
|
219
|
+
*/
|
|
220
|
+
export function buildBasenameLookup(entries) {
|
|
221
|
+
const lookup = new Map();
|
|
222
|
+
for (const entry of entries) {
|
|
223
|
+
const basename = entry.file
|
|
224
|
+
.split('/')
|
|
225
|
+
.pop()
|
|
226
|
+
.replace(/\.mdx?$/i, '')
|
|
227
|
+
.toLowerCase();
|
|
228
|
+
const targets = lookup.get(basename) ?? [];
|
|
229
|
+
targets.push({ slug: entry.slug, collection: entry.collection, urlPath: entry.urlPath });
|
|
230
|
+
lookup.set(basename, targets);
|
|
231
|
+
}
|
|
232
|
+
return lookup;
|
|
233
|
+
}
|
|
234
|
+
/**
|
|
235
|
+
* Resolve one extracted edge, using the table its kind belongs to.
|
|
236
|
+
*
|
|
237
|
+
* Deliberately does nothing clever: no basename fallback, no source-relative
|
|
238
|
+
* disambiguation, no collision policy. Those are resolution questions, tracked
|
|
239
|
+
* on #18; this is extraction's other half and nothing more.
|
|
240
|
+
*/
|
|
241
|
+
export function resolveLink(link, byName, byUrl) {
|
|
242
|
+
return link.kind === 'url'
|
|
243
|
+
? byUrl.get(link.target)
|
|
244
|
+
: byName.get(link.target.toLowerCase());
|
|
245
|
+
}
|
|
246
|
+
/**
|
|
247
|
+
* One in-flight or settled lookup per project root, keyed by its absolute path.
|
|
248
|
+
*
|
|
249
|
+
* Keyed on the *root* rather than built once per process, because the process
|
|
250
|
+
* is not the unit of work any more: the package is installed into somebody
|
|
251
|
+
* else's project, and the remark plugin there has to resolve `[[WikiLinks]]`
|
|
252
|
+
* against *their* content tree. A single module-level lookup would be filled by
|
|
253
|
+
* whichever project scanned first and then silently answered for every other
|
|
254
|
+
* one — the same failure `loadContentEntries` avoids by taking a root instead
|
|
255
|
+
* of calling `process.chdir`.
|
|
256
|
+
*
|
|
257
|
+
* The value is the promise, not the map, so the concurrent transforms of one
|
|
258
|
+
* build share a single scan rather than racing to start several. A rejected
|
|
259
|
+
* scan is evicted: a failure is a moment, not an answer, and remembering it
|
|
260
|
+
* would make one bad read permanent for the life of the process.
|
|
261
|
+
*/
|
|
262
|
+
const lookupCache = new Map();
|
|
263
|
+
/**
|
|
264
|
+
* The link lookup for one project root, built once and reused.
|
|
265
|
+
*
|
|
266
|
+
* The remark plugin runs per markdown file, so rescanning the content tree on
|
|
267
|
+
* every transform would make builds quadratic. Call `resetGraphCache()` if a
|
|
268
|
+
* caller needs to see content written during the same process.
|
|
269
|
+
*/
|
|
270
|
+
export function getLinkLookup(options = {}) {
|
|
271
|
+
// Resolved, so `.`, `./`, a relative path and the absolute one are one entry
|
|
272
|
+
// rather than four scans of the same tree.
|
|
273
|
+
const root = path.resolve(options.root ?? process.cwd());
|
|
274
|
+
const cached = lookupCache.get(root);
|
|
275
|
+
if (cached)
|
|
276
|
+
return cached;
|
|
277
|
+
const pending = loadContentEntries({ root }).then(buildLinkLookup);
|
|
278
|
+
pending.catch(() => lookupCache.delete(root));
|
|
279
|
+
lookupCache.set(root, pending);
|
|
280
|
+
return pending;
|
|
281
|
+
}
|
|
282
|
+
export function resetGraphCache() {
|
|
283
|
+
lookupCache.clear();
|
|
284
|
+
}
|
|
285
|
+
/**
|
|
286
|
+
* Blank out fenced blocks and inline code, preserving offsets.
|
|
287
|
+
*
|
|
288
|
+
* A `[[WikiLink]]` written inside backticks is documentation *about* the
|
|
289
|
+
* syntax, not a link. The remark plugin gets this right for free because it
|
|
290
|
+
* only visits `text` nodes and never `inlineCode` — anything scanning raw
|
|
291
|
+
* markdown has to strip code itself or it reports phantom broken links.
|
|
292
|
+
*/
|
|
293
|
+
export function stripCode(content) {
|
|
294
|
+
return content
|
|
295
|
+
.replace(/```[\s\S]*?```/g, (block) => block.replace(/[^\n]/g, ' '))
|
|
296
|
+
.replace(/~~~[\s\S]*?~~~/g, (block) => block.replace(/[^\n]/g, ' '))
|
|
297
|
+
.replace(/`[^`\n]*`/g, (span) => ' '.repeat(span.length));
|
|
298
|
+
}
|
|
299
|
+
/** Frontmatter keys whose values are vocabulary, not links. */
|
|
300
|
+
const NON_LINK_KEYS = new Set(['aliases', 'tags']);
|
|
301
|
+
/**
|
|
302
|
+
* Extract every outbound link target from a markdown body and its frontmatter.
|
|
303
|
+
*
|
|
304
|
+
* Returns the edges Obsidian's `resolvedLinks` would record: wikilinks, embeds,
|
|
305
|
+
* internal markdown links, and `frontmatterLinks`. Code is excluded.
|
|
306
|
+
*
|
|
307
|
+
* Frontmatter is passed separately because it is already parsed by the time it
|
|
308
|
+
* reaches here — `loadContentEntries()` splits it off with gray-matter, so the
|
|
309
|
+
* body alone can never contain it.
|
|
310
|
+
*/
|
|
311
|
+
export function extractLinks(content, frontmatter = {}) {
|
|
312
|
+
const links = [];
|
|
313
|
+
const prose = stripCode(content);
|
|
314
|
+
for (const match of prose.matchAll(WIKILINK)) {
|
|
315
|
+
const target = stripSubpath(match[1]);
|
|
316
|
+
if (target)
|
|
317
|
+
links.push({ kind: 'name', target });
|
|
318
|
+
}
|
|
319
|
+
for (const match of prose.matchAll(MARKDOWN_LINK)) {
|
|
320
|
+
const link = classifyMarkdownTarget(match[1]);
|
|
321
|
+
if (link)
|
|
322
|
+
links.push(link);
|
|
323
|
+
}
|
|
324
|
+
links.push(...extractFrontmatterLinks(frontmatter));
|
|
325
|
+
return dedupe(links);
|
|
326
|
+
}
|
|
327
|
+
/**
|
|
328
|
+
* Wikilinks reachable from frontmatter values, at any nesting depth.
|
|
329
|
+
*
|
|
330
|
+
* Only wikilinks: a frontmatter value is not markdown, and Obsidian's
|
|
331
|
+
* `frontmatterLinks` records the wikilink form. `aliases` and `tags` are names
|
|
332
|
+
* this entry answers to, not links out of it.
|
|
333
|
+
*/
|
|
334
|
+
function extractFrontmatterLinks(frontmatter) {
|
|
335
|
+
const links = [];
|
|
336
|
+
const walk = (value) => {
|
|
337
|
+
if (typeof value === 'string') {
|
|
338
|
+
for (const match of value.matchAll(WIKILINK)) {
|
|
339
|
+
const target = stripSubpath(match[1]);
|
|
340
|
+
if (target)
|
|
341
|
+
links.push({ kind: 'name', target });
|
|
342
|
+
}
|
|
343
|
+
}
|
|
344
|
+
else if (Array.isArray(value)) {
|
|
345
|
+
value.forEach(walk);
|
|
346
|
+
}
|
|
347
|
+
else if (value && typeof value === 'object') {
|
|
348
|
+
for (const [key, nested] of Object.entries(value)) {
|
|
349
|
+
if (!NON_LINK_KEYS.has(key))
|
|
350
|
+
walk(nested);
|
|
351
|
+
}
|
|
352
|
+
}
|
|
353
|
+
};
|
|
354
|
+
for (const [key, value] of Object.entries(frontmatter)) {
|
|
355
|
+
if (!NON_LINK_KEYS.has(key))
|
|
356
|
+
walk(value);
|
|
357
|
+
}
|
|
358
|
+
return links;
|
|
359
|
+
}
|
|
360
|
+
/** A wikilink or embed, capturing the target and discarding any display text. */
|
|
361
|
+
const WIKILINK = /\[\[([^\]|]+)(?:\|[^\]]+)?\]\]/g;
|
|
362
|
+
/**
|
|
363
|
+
* A markdown link or image, capturing the destination.
|
|
364
|
+
*
|
|
365
|
+
* The destination is either `<angle bracketed>` or unspaced, and may be
|
|
366
|
+
* followed by a title in quotes. Link text is `[^\]]*` rather than `[^\]]+`
|
|
367
|
+
* because an image embed can legitimately carry no alt text.
|
|
368
|
+
*/
|
|
369
|
+
const MARKDOWN_LINK = /!?\[[^\]]*\]\(\s*(<[^>]*>|[^()\s]+)(?:\s+"[^"]*")?\s*\)/g;
|
|
370
|
+
/**
|
|
371
|
+
* Turn a markdown link destination into an edge, or nothing if it leaves the site.
|
|
372
|
+
*
|
|
373
|
+
* Absolute paths are url-shaped and are normalised to the canonical spelling
|
|
374
|
+
* the graph stores — query, fragment, and the presence or absence of a trailing
|
|
375
|
+
* slash are all spellings of one edge, and `/notes/atomic-notes` must not be
|
|
376
|
+
* handed to the title lookup, where it only ever resolved by the coincidence of
|
|
377
|
+
* a note listing its own slug as an alias.
|
|
378
|
+
*/
|
|
379
|
+
function classifyMarkdownTarget(raw) {
|
|
380
|
+
const destination = raw.replace(/^<|>$/g, '').trim();
|
|
381
|
+
// Anything with a scheme, and protocol-relative `//host`, leaves the site.
|
|
382
|
+
if (!destination || /^[a-z][a-z0-9+.-]*:/i.test(destination))
|
|
383
|
+
return null;
|
|
384
|
+
if (destination.startsWith('//'))
|
|
385
|
+
return null;
|
|
386
|
+
if (destination.startsWith('/')) {
|
|
387
|
+
const pathname = destination.split(/[?#]/)[0];
|
|
388
|
+
if (pathname === '/')
|
|
389
|
+
return null;
|
|
390
|
+
return { kind: 'url', target: pathname.endsWith('/') ? pathname : `${pathname}/` };
|
|
391
|
+
}
|
|
392
|
+
// A relative link to a markdown file — what Obsidian writes when a note is
|
|
393
|
+
// dragged into another. Only the basename carries meaning here: the target
|
|
394
|
+
// resolves by title, exactly as the wikilink spelling of the same edge does.
|
|
395
|
+
// Which directory it came from is a resolution question, tracked on #18.
|
|
396
|
+
const file = decodePath(stripSubpath(destination));
|
|
397
|
+
if (!/\.mdx?$/i.test(file))
|
|
398
|
+
return null;
|
|
399
|
+
const title = file.split('/').pop().replace(/\.mdx?$/i, '');
|
|
400
|
+
return title ? { kind: 'name', target: title } : null;
|
|
401
|
+
}
|
|
402
|
+
/** Percent-decode a link destination, leaving a malformed one alone. */
|
|
403
|
+
function decodePath(destination) {
|
|
404
|
+
try {
|
|
405
|
+
return decodeURIComponent(destination);
|
|
406
|
+
}
|
|
407
|
+
catch {
|
|
408
|
+
return destination;
|
|
409
|
+
}
|
|
410
|
+
}
|
|
411
|
+
/** Drop an Obsidian subpath: everything from the first `#` heading or `^` block id.
|
|
412
|
+
*
|
|
413
|
+
* `resolvedLinks` records `[[Note#Heading]]` as an edge to `Note`, so the
|
|
414
|
+
* subpath must not survive into the target. What remains can be empty — a bare
|
|
415
|
+
* `[[#Heading]]` points inside the current file and is not an edge at all.
|
|
416
|
+
*/
|
|
417
|
+
function stripSubpath(text) {
|
|
418
|
+
return text.split(/[#^]/)[0].trim();
|
|
419
|
+
}
|
|
420
|
+
/** Drop repeated edges, keeping first-seen order. */
|
|
421
|
+
function dedupe(links) {
|
|
422
|
+
const seen = new Set();
|
|
423
|
+
return links.filter((link) => {
|
|
424
|
+
const key = `${link.kind}:${link.target}`;
|
|
425
|
+
if (seen.has(key))
|
|
426
|
+
return false;
|
|
427
|
+
seen.add(key);
|
|
428
|
+
return true;
|
|
429
|
+
});
|
|
430
|
+
}
|
|
431
|
+
/**
|
|
432
|
+
* Star calculation strategy.
|
|
433
|
+
*
|
|
434
|
+
* Lives in the core rather than the Astro integration because `isStarred` is a
|
|
435
|
+
* field of the public artifact, and the CLI has to be able to produce the same
|
|
436
|
+
* artifact without loading Astro.
|
|
437
|
+
*/
|
|
438
|
+
export const STAR_CONFIG = {
|
|
439
|
+
strategy: 'top-percent',
|
|
440
|
+
topPercent: 5, // Top 5%
|
|
441
|
+
topAbsolute: 3, // Top 3 notes
|
|
442
|
+
threshold: 10, // 10+ backlinks
|
|
443
|
+
minNotesForStars: 20,
|
|
444
|
+
rankBy: 'backlinks',
|
|
445
|
+
weights: {
|
|
446
|
+
backlinks: 0.5,
|
|
447
|
+
revisions: 0.3,
|
|
448
|
+
crossTheme: 0.2,
|
|
449
|
+
},
|
|
450
|
+
};
|
|
451
|
+
/**
|
|
452
|
+
* Calculate which notes get stars based on STAR_CONFIG.
|
|
453
|
+
* Returns a Set of slugs that should be starred.
|
|
454
|
+
*/
|
|
455
|
+
export function calculateStars(notes) {
|
|
456
|
+
const starredSlugs = new Set();
|
|
457
|
+
// Standalone pages belong in search and WikiLinks, not note rankings.
|
|
458
|
+
const notesArray = Array.from(notes.values()).filter((note) => note.collection !== 'pages');
|
|
459
|
+
// Skip if too few notes
|
|
460
|
+
if (notesArray.length < STAR_CONFIG.minNotesForStars) {
|
|
461
|
+
return starredSlugs;
|
|
462
|
+
}
|
|
463
|
+
// Calculate score based on rankBy strategy
|
|
464
|
+
const scored = notesArray.map((note) => {
|
|
465
|
+
let score = 0;
|
|
466
|
+
switch (STAR_CONFIG.rankBy) {
|
|
467
|
+
case 'backlinks':
|
|
468
|
+
score = note.inbound.length;
|
|
469
|
+
break;
|
|
470
|
+
case 'revisions':
|
|
471
|
+
// Future: could track revision count in frontmatter or git history
|
|
472
|
+
score = 0;
|
|
473
|
+
break;
|
|
474
|
+
case 'cross-theme':
|
|
475
|
+
// Future: count unique tags across linked notes
|
|
476
|
+
score = 0;
|
|
477
|
+
break;
|
|
478
|
+
case 'weighted':
|
|
479
|
+
// Future: combine multiple metrics
|
|
480
|
+
score = note.inbound.length * STAR_CONFIG.weights.backlinks;
|
|
481
|
+
break;
|
|
482
|
+
}
|
|
483
|
+
return { slug: note.slug, score };
|
|
484
|
+
});
|
|
485
|
+
// Sort by score descending
|
|
486
|
+
scored.sort((a, b) => b.score - a.score);
|
|
487
|
+
// Determine which notes get stars based on strategy
|
|
488
|
+
let cutoffIndex = 0;
|
|
489
|
+
switch (STAR_CONFIG.strategy) {
|
|
490
|
+
case 'top-percent':
|
|
491
|
+
cutoffIndex = Math.ceil(notesArray.length * (STAR_CONFIG.topPercent / 100));
|
|
492
|
+
break;
|
|
493
|
+
case 'top-absolute':
|
|
494
|
+
cutoffIndex = Math.min(STAR_CONFIG.topAbsolute, notesArray.length);
|
|
495
|
+
break;
|
|
496
|
+
case 'threshold':
|
|
497
|
+
// All notes above threshold get stars
|
|
498
|
+
for (const { slug, score } of scored) {
|
|
499
|
+
if (score >= STAR_CONFIG.threshold) {
|
|
500
|
+
starredSlugs.add(slug);
|
|
501
|
+
}
|
|
502
|
+
}
|
|
503
|
+
return starredSlugs;
|
|
504
|
+
}
|
|
505
|
+
// Add top N notes to starred set
|
|
506
|
+
for (let i = 0; i < cutoffIndex; i++) {
|
|
507
|
+
starredSlugs.add(scored[i].slug);
|
|
508
|
+
}
|
|
509
|
+
// Handle ties at boundary
|
|
510
|
+
if (cutoffIndex > 0 && cutoffIndex < scored.length) {
|
|
511
|
+
const boundaryScore = scored[cutoffIndex - 1].score;
|
|
512
|
+
// Include all notes tied with the boundary score
|
|
513
|
+
for (let i = cutoffIndex; i < scored.length; i++) {
|
|
514
|
+
if (scored[i].score === boundaryScore) {
|
|
515
|
+
starredSlugs.add(scored[i].slug);
|
|
516
|
+
}
|
|
517
|
+
else {
|
|
518
|
+
break;
|
|
519
|
+
}
|
|
520
|
+
}
|
|
521
|
+
}
|
|
522
|
+
return starredSlugs;
|
|
523
|
+
}
|
|
524
|
+
/**
|
|
525
|
+
* Build the resolved graph from loaded entries.
|
|
526
|
+
*
|
|
527
|
+
* Pure: no filesystem, no logging, no Astro. Everything a caller might want to
|
|
528
|
+
* print comes back in `diagnostics`, so the same construction serves the build
|
|
529
|
+
* (which warns), `check` (which reports) and `related` (which does neither).
|
|
530
|
+
*/
|
|
531
|
+
export function buildGraph(entries) {
|
|
532
|
+
const byName = buildLinkLookup(entries);
|
|
533
|
+
const byUrl = buildUrlLookup(entries);
|
|
534
|
+
const byBasename = buildBasenameLookup(entries);
|
|
535
|
+
const notes = new Map();
|
|
536
|
+
// Raw extracted edges, kept beside the graph: `outbound` on NoteMetadata is
|
|
537
|
+
// the public artifact and holds resolved urlPaths only.
|
|
538
|
+
const extracted = new Map();
|
|
539
|
+
const files = new Map();
|
|
540
|
+
for (const entry of entries) {
|
|
541
|
+
extracted.set(entry.urlPath, extractLinks(entry.body, entry.frontmatter));
|
|
542
|
+
files.set(entry.urlPath, entry.file);
|
|
543
|
+
notes.set(entry.urlPath, {
|
|
544
|
+
slug: entry.urlPath,
|
|
545
|
+
title: entry.title,
|
|
546
|
+
collection: entry.collection,
|
|
547
|
+
aliases: entry.aliases,
|
|
548
|
+
outbound: [], // populated below, once links resolve
|
|
549
|
+
inbound: [], // populated below
|
|
550
|
+
tags: entry.tags,
|
|
551
|
+
status: entry.status,
|
|
552
|
+
...(entry.summary ? { summary: entry.summary } : {}),
|
|
553
|
+
...(entry.updated ? { updated: entry.updated } : {}),
|
|
554
|
+
});
|
|
555
|
+
}
|
|
556
|
+
const diagnostics = [];
|
|
557
|
+
// Resolve WikiLinks and compute inbound links
|
|
558
|
+
for (const [fromUrl, note] of notes.entries()) {
|
|
559
|
+
const resolvedOutbound = [];
|
|
560
|
+
for (const link of extracted.get(fromUrl)) {
|
|
561
|
+
const resolved = resolveLink(link, byName, byUrl)?.urlPath;
|
|
562
|
+
// A `name` link that means one entry by title and a different one by
|
|
563
|
+
// filename resolves — it just may not resolve to what the author saw
|
|
564
|
+
// in Obsidian. Reported, not repaired: picking a winner here would
|
|
565
|
+
// make the two tools disagree quietly instead of loudly.
|
|
566
|
+
if (link.kind === 'name') {
|
|
567
|
+
const byFile = byBasename.get(link.target.toLowerCase());
|
|
568
|
+
if (resolved && byFile?.length === 1 && byFile[0].urlPath !== resolved) {
|
|
569
|
+
diagnostics.push({
|
|
570
|
+
rule: 'ambiguous-target',
|
|
571
|
+
severity: 'error',
|
|
572
|
+
file: files.get(fromUrl),
|
|
573
|
+
urlPath: fromUrl,
|
|
574
|
+
kind: link.kind,
|
|
575
|
+
target: link.target,
|
|
576
|
+
candidates: [resolved, byFile[0].urlPath].sort(),
|
|
577
|
+
message: `${spellLink(link)} resolves by title to ${resolved} but names the file ${byFile[0].urlPath}`,
|
|
578
|
+
});
|
|
579
|
+
}
|
|
580
|
+
}
|
|
581
|
+
if (resolved && notes.has(resolved)) {
|
|
582
|
+
resolvedOutbound.push(resolved);
|
|
583
|
+
const target = notes.get(resolved);
|
|
584
|
+
if (!target.inbound.includes(fromUrl)) {
|
|
585
|
+
target.inbound.push(fromUrl);
|
|
586
|
+
}
|
|
587
|
+
}
|
|
588
|
+
else {
|
|
589
|
+
// Link to a note that doesn't exist yet — fine, but worth saying
|
|
590
|
+
diagnostics.push({
|
|
591
|
+
rule: 'broken-link',
|
|
592
|
+
severity: 'warning',
|
|
593
|
+
file: files.get(fromUrl),
|
|
594
|
+
urlPath: fromUrl,
|
|
595
|
+
kind: link.kind,
|
|
596
|
+
target: link.target,
|
|
597
|
+
message: `${spellLink(link)} does not resolve`,
|
|
598
|
+
});
|
|
599
|
+
}
|
|
600
|
+
}
|
|
601
|
+
note.outbound = resolvedOutbound;
|
|
602
|
+
}
|
|
603
|
+
// Calculate which notes get stars
|
|
604
|
+
const starredSlugs = calculateStars(notes);
|
|
605
|
+
for (const slug of starredSlugs) {
|
|
606
|
+
const note = notes.get(slug);
|
|
607
|
+
if (note) {
|
|
608
|
+
note.isStarred = true;
|
|
609
|
+
}
|
|
610
|
+
}
|
|
611
|
+
const nodes = {};
|
|
612
|
+
for (const [slug, note] of notes.entries()) {
|
|
613
|
+
nodes[slug] = note;
|
|
614
|
+
}
|
|
615
|
+
const totalBacklinks = Array.from(notes.values()).reduce((sum, note) => sum + note.inbound.length, 0);
|
|
616
|
+
return { nodes, diagnostics, totalBacklinks };
|
|
617
|
+
}
|
|
618
|
+
/** How a link was written, in the spelling its kind uses. */
|
|
619
|
+
function spellLink(link) {
|
|
620
|
+
return link.kind === 'url' ? `(${link.target})` : `[[${link.target}]]`;
|
|
621
|
+
}
|
|
622
|
+
/**
|
|
623
|
+
* Render a diagnostic as the build has always rendered it.
|
|
624
|
+
*
|
|
625
|
+
* `public/backlinks.json` is not the only committed contract — build output is
|
|
626
|
+
* read by humans and diffed by reviewers, so the warning line is kept exactly
|
|
627
|
+
* as it was when the integration formatted it itself.
|
|
628
|
+
*/
|
|
629
|
+
export function formatDiagnostic(diagnostic) {
|
|
630
|
+
if (diagnostic.rule === 'broken-link') {
|
|
631
|
+
return `⚠️ Broken link in ${diagnostic.urlPath}: ${spellLink({
|
|
632
|
+
kind: diagnostic.kind ?? 'name',
|
|
633
|
+
target: diagnostic.target ?? '',
|
|
634
|
+
})}`;
|
|
635
|
+
}
|
|
636
|
+
return `⚠️ ${diagnostic.rule} in ${diagnostic.file}: ${diagnostic.message}`;
|
|
637
|
+
}
|
|
638
|
+
/**
|
|
639
|
+
* The public artifact, exactly as `public/backlinks.json` stores it.
|
|
640
|
+
*
|
|
641
|
+
* Separate from `Graph` because the file is a runtime contract with four
|
|
642
|
+
* client-side readers: the graph can grow fields, the file cannot.
|
|
643
|
+
*/
|
|
644
|
+
export function toBacklinksJson(graph) {
|
|
645
|
+
return graph.nodes;
|
|
646
|
+
}
|
|
647
|
+
/**
|
|
648
|
+
* Names claimed by more than one entry.
|
|
649
|
+
*
|
|
650
|
+
* Two triggers, because a name lives in two namespaces: a title or alias that
|
|
651
|
+
* two entries both answer to, and a filename that two entries share. Neither
|
|
652
|
+
* is an error the graph can resolve — `buildLinkLookup` keeps last-wins, which
|
|
653
|
+
* is what it has always done — but both mean a `[[WikiLink]]` reaches somewhere
|
|
654
|
+
* its author did not choose, and silently.
|
|
655
|
+
*
|
|
656
|
+
* Reported against the *first* claimant, because that is the entry the
|
|
657
|
+
* collision makes unreachable.
|
|
658
|
+
*/
|
|
659
|
+
export function findDuplicateNames(entries) {
|
|
660
|
+
const byTitle = new Map();
|
|
661
|
+
const byBasename = new Map();
|
|
662
|
+
for (const entry of entries) {
|
|
663
|
+
for (const name of [entry.title, ...entry.aliases]) {
|
|
664
|
+
const key = name.toLowerCase();
|
|
665
|
+
byTitle.set(key, [...(byTitle.get(key) ?? []), { entry, name }]);
|
|
666
|
+
}
|
|
667
|
+
const basename = entry.file.split('/').pop().replace(/\.mdx?$/i, '').toLowerCase();
|
|
668
|
+
byBasename.set(basename, [...(byBasename.get(basename) ?? []), entry]);
|
|
669
|
+
}
|
|
670
|
+
const diagnostics = [];
|
|
671
|
+
for (const [, all] of byBasename) {
|
|
672
|
+
const claimants = distinct(all, (entry) => entry.urlPath);
|
|
673
|
+
if (claimants.length < 2)
|
|
674
|
+
continue;
|
|
675
|
+
const [first] = claimants;
|
|
676
|
+
diagnostics.push({
|
|
677
|
+
rule: 'duplicate-name',
|
|
678
|
+
severity: 'error',
|
|
679
|
+
file: first.file,
|
|
680
|
+
target: first.file.split('/').pop().replace(/\.mdx?$/i, ''),
|
|
681
|
+
candidates: claimants.map((entry) => entry.urlPath).sort(),
|
|
682
|
+
message: `${claimants.length} entries share the filename ${first.file
|
|
683
|
+
.split('/')
|
|
684
|
+
.pop()}: ${claimants.map((entry) => entry.urlPath).join(', ')}`,
|
|
685
|
+
});
|
|
686
|
+
}
|
|
687
|
+
for (const [, all] of byTitle) {
|
|
688
|
+
// One entry may answer to a name twice — a note titled `Commune` that
|
|
689
|
+
// also lists `commune` as an alias is spelling the same claim twice, not
|
|
690
|
+
// competing with itself.
|
|
691
|
+
const claimants = distinct(all, ({ entry }) => entry.urlPath);
|
|
692
|
+
if (claimants.length < 2)
|
|
693
|
+
continue;
|
|
694
|
+
const [first] = claimants;
|
|
695
|
+
diagnostics.push({
|
|
696
|
+
rule: 'duplicate-name',
|
|
697
|
+
severity: 'error',
|
|
698
|
+
file: first.entry.file,
|
|
699
|
+
target: first.name,
|
|
700
|
+
candidates: claimants.map(({ entry }) => entry.urlPath).sort(),
|
|
701
|
+
message: `${claimants.length} entries answer to "${first.name}": ${claimants
|
|
702
|
+
.map(({ entry }) => entry.urlPath)
|
|
703
|
+
.join(', ')} — the last one wins`,
|
|
704
|
+
});
|
|
705
|
+
}
|
|
706
|
+
return diagnostics;
|
|
707
|
+
}
|
|
708
|
+
/** Keep the first item per key, so an entry cannot collide with itself. */
|
|
709
|
+
function distinct(items, key) {
|
|
710
|
+
const seen = new Set();
|
|
711
|
+
return items.filter((item) => {
|
|
712
|
+
const id = key(item);
|
|
713
|
+
if (seen.has(id))
|
|
714
|
+
return false;
|
|
715
|
+
seen.add(id);
|
|
716
|
+
return true;
|
|
717
|
+
});
|
|
718
|
+
}
|
|
719
|
+
/**
|
|
720
|
+
* WikiLinks that resolve but do not name their target exactly.
|
|
721
|
+
*
|
|
722
|
+
* The rule this implements used to live in the build gate script, with its own
|
|
723
|
+
* lookup table and its own regex — a second copy of a rule the graph core
|
|
724
|
+
* already had the ingredients for, which is the bug class #3 was opened to
|
|
725
|
+
* kill. One implementation now, two callers: `check` reports it and
|
|
726
|
+
* `commune gate` fails on it.
|
|
727
|
+
*
|
|
728
|
+
* Worth checking precisely because it is invisible: a piped link or a
|
|
729
|
+
* near-miss title still renders, and the vault quietly decouples from the site.
|
|
730
|
+
*/
|
|
731
|
+
export function findNoncanonicalTitles(entries) {
|
|
732
|
+
const canonical = new Map();
|
|
733
|
+
for (const entry of entries) {
|
|
734
|
+
canonical.set(entry.title.toLowerCase(), entry.title);
|
|
735
|
+
for (const alias of entry.aliases) {
|
|
736
|
+
canonical.set(alias.toLowerCase(), entry.title);
|
|
737
|
+
}
|
|
738
|
+
}
|
|
739
|
+
const diagnostics = [];
|
|
740
|
+
for (const entry of entries) {
|
|
741
|
+
for (const match of stripCode(entry.body).matchAll(LABELLED_WIKILINK)) {
|
|
742
|
+
const linked = match[1].trim();
|
|
743
|
+
const label = match[2]?.trim();
|
|
744
|
+
const exact = canonical.get(linked.toLowerCase());
|
|
745
|
+
if (!exact)
|
|
746
|
+
continue; // unresolved links are `broken-link`'s business
|
|
747
|
+
if (label || linked !== exact) {
|
|
748
|
+
diagnostics.push({
|
|
749
|
+
rule: 'noncanonical-title',
|
|
750
|
+
severity: 'error',
|
|
751
|
+
file: entry.file,
|
|
752
|
+
urlPath: entry.urlPath,
|
|
753
|
+
kind: 'name',
|
|
754
|
+
target: exact,
|
|
755
|
+
canonical: exact,
|
|
756
|
+
message: `[[${linked}${label ? `|${label}` : ''}]] should be [[${exact}]]`,
|
|
757
|
+
});
|
|
758
|
+
}
|
|
759
|
+
}
|
|
760
|
+
}
|
|
761
|
+
return diagnostics;
|
|
762
|
+
}
|
|
763
|
+
/** A wikilink keeping its display text, which the canonical-title rule needs to see. */
|
|
764
|
+
const LABELLED_WIKILINK = /\[\[([^\]|]+)(?:\|([^\]]+))?\]\]/g;
|
|
765
|
+
/**
|
|
766
|
+
* Every finding `check` reports.
|
|
767
|
+
*
|
|
768
|
+
* The graph's own diagnostics come first and in entry order, so the list reads
|
|
769
|
+
* the same way the build log does, then the two whole-corpus rules that need
|
|
770
|
+
* every entry in hand before they can fire.
|
|
771
|
+
*/
|
|
772
|
+
export function checkEntries(entries, graph) {
|
|
773
|
+
return [...graph.diagnostics, ...findDuplicateNames(entries), ...findNoncanonicalTitles(entries)];
|
|
774
|
+
}
|