obsidian-mcp-server 3.5.4 → 3.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +9 -3
- package/CLAUDE.md +9 -3
- package/README.md +14 -13
- package/changelog/3.5.x/3.5.5.md +20 -0
- package/changelog/3.6.x/3.6.0.md +31 -0
- package/dist/mcp-server/tools/definitions/_shared/schemas.d.ts.map +1 -1
- package/dist/mcp-server/tools/definitions/_shared/schemas.js +2 -2
- package/dist/mcp-server/tools/definitions/_shared/schemas.js.map +1 -1
- package/dist/mcp-server/tools/definitions/index.d.ts +97 -61
- package/dist/mcp-server/tools/definitions/index.d.ts.map +1 -1
- package/dist/mcp-server/tools/definitions/obsidian-append-to-note.tool.d.ts +14 -2
- package/dist/mcp-server/tools/definitions/obsidian-append-to-note.tool.d.ts.map +1 -1
- package/dist/mcp-server/tools/definitions/obsidian-append-to-note.tool.js +17 -4
- package/dist/mcp-server/tools/definitions/obsidian-append-to-note.tool.js.map +1 -1
- package/dist/mcp-server/tools/definitions/obsidian-get-note.tool.d.ts.map +1 -1
- package/dist/mcp-server/tools/definitions/obsidian-get-note.tool.js +4 -3
- package/dist/mcp-server/tools/definitions/obsidian-get-note.tool.js.map +1 -1
- package/dist/mcp-server/tools/definitions/obsidian-manage-tags.tool.d.ts +5 -2
- package/dist/mcp-server/tools/definitions/obsidian-manage-tags.tool.d.ts.map +1 -1
- package/dist/mcp-server/tools/definitions/obsidian-manage-tags.tool.js +6 -3
- package/dist/mcp-server/tools/definitions/obsidian-manage-tags.tool.js.map +1 -1
- package/dist/mcp-server/tools/definitions/obsidian-patch-note.tool.d.ts +17 -4
- package/dist/mcp-server/tools/definitions/obsidian-patch-note.tool.d.ts.map +1 -1
- package/dist/mcp-server/tools/definitions/obsidian-patch-note.tool.js +20 -6
- package/dist/mcp-server/tools/definitions/obsidian-patch-note.tool.js.map +1 -1
- package/dist/mcp-server/tools/definitions/obsidian-search-notes.tool.d.ts +6 -5
- package/dist/mcp-server/tools/definitions/obsidian-search-notes.tool.d.ts.map +1 -1
- package/dist/mcp-server/tools/definitions/obsidian-search-notes.tool.js +26 -24
- package/dist/mcp-server/tools/definitions/obsidian-search-notes.tool.js.map +1 -1
- package/dist/mcp-server/tools/definitions/obsidian-write-note.tool.d.ts +14 -2
- package/dist/mcp-server/tools/definitions/obsidian-write-note.tool.d.ts.map +1 -1
- package/dist/mcp-server/tools/definitions/obsidian-write-note.tool.js +26 -13
- package/dist/mcp-server/tools/definitions/obsidian-write-note.tool.js.map +1 -1
- package/dist/services/obsidian/frontmatter-ops.d.ts +7 -4
- package/dist/services/obsidian/frontmatter-ops.d.ts.map +1 -1
- package/dist/services/obsidian/frontmatter-ops.js +502 -59
- package/dist/services/obsidian/frontmatter-ops.js.map +1 -1
- package/dist/services/obsidian/markdown-blocks.d.ts +39 -0
- package/dist/services/obsidian/markdown-blocks.d.ts.map +1 -0
- package/dist/services/obsidian/markdown-blocks.js +611 -0
- package/dist/services/obsidian/markdown-blocks.js.map +1 -0
- package/dist/services/obsidian/obsidian-service.d.ts +18 -12
- package/dist/services/obsidian/obsidian-service.d.ts.map +1 -1
- package/dist/services/obsidian/obsidian-service.js +653 -130
- package/dist/services/obsidian/obsidian-service.js.map +1 -1
- package/dist/services/obsidian/patch-instruction.d.ts +141 -0
- package/dist/services/obsidian/patch-instruction.d.ts.map +1 -0
- package/dist/services/obsidian/patch-instruction.js +217 -0
- package/dist/services/obsidian/patch-instruction.js.map +1 -0
- package/dist/services/obsidian/section-extractor.d.ts +109 -6
- package/dist/services/obsidian/section-extractor.d.ts.map +1 -1
- package/dist/services/obsidian/section-extractor.js +364 -87
- package/dist/services/obsidian/section-extractor.js.map +1 -1
- package/dist/services/obsidian/types.d.ts +23 -9
- package/dist/services/obsidian/types.d.ts.map +1 -1
- package/manifest.json +1 -1
- package/package.json +3 -2
- package/server.json +3 -3
|
@@ -5,6 +5,7 @@
|
|
|
5
5
|
* @module services/obsidian/frontmatter-ops
|
|
6
6
|
*/
|
|
7
7
|
import { isMap, isScalar, isSeq, parseDocument, Scalar } from 'yaml';
|
|
8
|
+
import { HTML_TAG_SOURCE, scanBlocks } from './markdown-blocks.js';
|
|
8
9
|
/**
|
|
9
10
|
* The one frontmatter boundary: an opening `---` alone on the first line, YAML,
|
|
10
11
|
* then a `---` that starts its own line. The YAML span and the newline that
|
|
@@ -164,8 +165,9 @@ export function deleteFrontmatterKey(content, key) {
|
|
|
164
165
|
}
|
|
165
166
|
/**
|
|
166
167
|
* Add or remove tags across frontmatter (`tags:` array) and inline `#tag`
|
|
167
|
-
* syntax. Inline detection skips code spans, link spans,
|
|
168
|
-
*
|
|
168
|
+
* syntax. Inline detection skips code spans, link spans, HTML comments, math,
|
|
169
|
+
* and a `#` preceded by anything but whitespace, line start, or markup such
|
|
170
|
+
* as `**` or `<br>` — see `splitProtectedSegments` and `TAG_LEFT_BOUNDARY`.
|
|
169
171
|
*/
|
|
170
172
|
export function reconcileTags(content, tags, operation, location) {
|
|
171
173
|
const norm = (t) => t.replace(/^#+/, '').trim();
|
|
@@ -308,8 +310,47 @@ function normalizeTagList(value) {
|
|
|
308
310
|
}
|
|
309
311
|
return [];
|
|
310
312
|
}
|
|
311
|
-
|
|
312
|
-
|
|
313
|
+
/**
|
|
314
|
+
* The opening backtick run of a code span. It closes at the next whole
|
|
315
|
+
* backtick run of the same length in the paragraph, across line breaks:
|
|
316
|
+
* `` a ``b ` #x`` `` is code, `` a `` #x ``` `` is not. Only the opener is a
|
|
317
|
+
* regex; `codeSpanCloser` finds the closer.
|
|
318
|
+
*/
|
|
319
|
+
const INLINE_CODE = /(?<code>`+)/;
|
|
320
|
+
/**
|
|
321
|
+
* The end of the code span each opener in `text` starts, or `undefined` when
|
|
322
|
+
* it has none. Whole backtick runs are listed once, by length; openers must be
|
|
323
|
+
* looked up in ascending order, so each length's list is walked once.
|
|
324
|
+
*/
|
|
325
|
+
function codeSpanCloser(text) {
|
|
326
|
+
const runs = new Map();
|
|
327
|
+
for (let i = 0; i < text.length;) {
|
|
328
|
+
if (text[i] !== '`') {
|
|
329
|
+
i++;
|
|
330
|
+
continue;
|
|
331
|
+
}
|
|
332
|
+
const start = i;
|
|
333
|
+
while (text[i] === '`')
|
|
334
|
+
i++;
|
|
335
|
+
const list = runs.get(i - start);
|
|
336
|
+
if (list)
|
|
337
|
+
list.push(start);
|
|
338
|
+
else
|
|
339
|
+
runs.set(i - start, [start]);
|
|
340
|
+
}
|
|
341
|
+
const walked = new Map();
|
|
342
|
+
return (open, length) => {
|
|
343
|
+
const list = runs.get(length);
|
|
344
|
+
if (!list)
|
|
345
|
+
return;
|
|
346
|
+
let k = walked.get(length) ?? 0;
|
|
347
|
+
while ((list[k] ?? Number.POSITIVE_INFINITY) <= open)
|
|
348
|
+
k++;
|
|
349
|
+
walked.set(length, k);
|
|
350
|
+
const close = list[k];
|
|
351
|
+
return close === undefined ? undefined : close + length;
|
|
352
|
+
};
|
|
353
|
+
}
|
|
313
354
|
/**
|
|
314
355
|
* `[[Target#Heading|Alias]]`. The `#` in a wikilink opens a heading anchor and
|
|
315
356
|
* the text after `|` is display text — neither is a tag. Protecting the whole
|
|
@@ -320,20 +361,184 @@ const INLINE_CODE = /`[^`\n]+`/;
|
|
|
320
361
|
* link target, so a span never nests.
|
|
321
362
|
*/
|
|
322
363
|
const WIKILINK = /\[\[[^[\]\n]*\]\]/;
|
|
323
|
-
/**
|
|
324
|
-
const
|
|
364
|
+
/** `` and `![alt][label]` — Obsidian reads no tag in an image's alt text. */
|
|
365
|
+
const IMAGE = /!\[[^[\]\n]*\](?:\([^()\n]*\)|\[[^[\]\n]*\])/;
|
|
366
|
+
/**
|
|
367
|
+
* The `](destination)` or `][label]` that makes the bracketed text before it a
|
|
368
|
+
* link. Only this tail is link syntax: Obsidian reads a tag in link text
|
|
369
|
+
* (`[Discord #tf](https://x.y)` is tagged `tf`) and none in a destination, a
|
|
370
|
+
* title, or a label. The text must open on the same line with no bracket in it;
|
|
371
|
+
* a bracketed phrase followed by a space and a second one (`[note #work] [ref]`)
|
|
372
|
+
* is no link.
|
|
373
|
+
*/
|
|
374
|
+
const LINK_TAIL = /(?<linkTail>\](?<=\[[^[\]\n]*\])(?:\([^()\n]*\)|\[[^[\]\n]*\]))/;
|
|
375
|
+
/** The `[label]:` that opens a link reference definition. */
|
|
376
|
+
const LINK_DEFINITION = /\[(?<=(?:^|\n)[ ]{0,3}\[)[^[\]\n]+\]:/;
|
|
377
|
+
/**
|
|
378
|
+
* Obsidian's metadata cache reads no tag inside an HTML comment or math. The
|
|
379
|
+
* rules below are pinned against its readback (Obsidian 1.13.7) rather than a
|
|
380
|
+
* spec. HTML blocks and display math blocks are block structure and live in
|
|
381
|
+
* `markdown-blocks.ts`; these are the spans inside one paragraph. Issue #138.
|
|
382
|
+
*
|
|
383
|
+
* An **inline comment**: `<!--`, then text that does not open with `>` or `->`
|
|
384
|
+
* and holds no `--`, then `-->`. `<!-- x -- y -->` is not a comment, and
|
|
385
|
+
* neither is an unclosed `<!--`.
|
|
386
|
+
*/
|
|
387
|
+
const HTML_COMMENT = /<!--(?!-?>)(?:(?!--|\n *\r?\n)[\s\S])*-->/;
|
|
388
|
+
/** Unescaped `$$ … $$` within one paragraph, with no spacing rule. */
|
|
389
|
+
const INLINE_DOUBLE_MATH = /(?<=(?:^|[^\\])(?:\\\\)*)\$\$(?:(?!\n *\r?\n)[\s\S])*?\$\$/;
|
|
390
|
+
/**
|
|
391
|
+
* Unescaped `$ … $` within one paragraph. The opener is not followed by a space
|
|
392
|
+
* or tab; the closer is not preceded by one and not followed by a digit, and a
|
|
393
|
+
* `$` that fails those is passed over rather than ending the span. That is what
|
|
394
|
+
* keeps `cost $5 and #rc for $10` out of math while `m $a #ra b$ n` is in it.
|
|
395
|
+
*
|
|
396
|
+
* Only the opener is a regex; `inlineMathCloser` finds the closer. A lazy
|
|
397
|
+
* regex body would rescan to the end of the paragraph from every opener that
|
|
398
|
+
* never closes — quadratic in a table of prices (issue #143).
|
|
399
|
+
*/
|
|
400
|
+
const INLINE_MATH_OPEN = /(?<inlineMath>(?<=(?:^|[^\\])(?:\\\\)*)\$(?![ \t$]))/;
|
|
401
|
+
/** The start of an empty or spaces-only line, which inline math cannot cross. */
|
|
402
|
+
const PARAGRAPH_BREAK = /\n *\r?\n/y;
|
|
403
|
+
/**
|
|
404
|
+
* The closer of the inline math span each opener in `content` starts, or
|
|
405
|
+
* `undefined` when it has none. Whether a `$` can close a span does not depend
|
|
406
|
+
* on where the span opened, so the closers and paragraph breaks are listed
|
|
407
|
+
* once and each opener takes the first closer after it, provided no break
|
|
408
|
+
* comes first. Openers must be looked up in ascending order.
|
|
409
|
+
*/
|
|
410
|
+
function inlineMathCloser(content) {
|
|
411
|
+
const closers = [];
|
|
412
|
+
const breaks = [];
|
|
413
|
+
for (let i = 0; i < content.length; i++) {
|
|
414
|
+
if (content[i] === '$' && closesInlineMath(content, i))
|
|
415
|
+
closers.push(i);
|
|
416
|
+
if (content[i] === '\n') {
|
|
417
|
+
PARAGRAPH_BREAK.lastIndex = i;
|
|
418
|
+
if (PARAGRAPH_BREAK.test(content))
|
|
419
|
+
breaks.push(i);
|
|
420
|
+
}
|
|
421
|
+
}
|
|
422
|
+
let c = 0;
|
|
423
|
+
let b = 0;
|
|
424
|
+
return (open) => {
|
|
425
|
+
while ((closers[c] ?? Infinity) <= open)
|
|
426
|
+
c++;
|
|
427
|
+
while ((breaks[b] ?? Infinity) <= open)
|
|
428
|
+
b++;
|
|
429
|
+
const close = closers[c];
|
|
430
|
+
if (close === undefined || (breaks[b] ?? Infinity) < close)
|
|
431
|
+
return;
|
|
432
|
+
return close;
|
|
433
|
+
};
|
|
434
|
+
}
|
|
435
|
+
/** Whether the `$` at `i` can close inline math: unescaped, after no space or tab, before no digit. */
|
|
436
|
+
function closesInlineMath(content, i) {
|
|
437
|
+
const before = content[i - 1];
|
|
438
|
+
if (before === ' ' || before === '\t' || /[0-9]/.test(content[i + 1] ?? ''))
|
|
439
|
+
return false;
|
|
440
|
+
let backslashes = 0;
|
|
441
|
+
while (content[i - 1 - backslashes] === '\\')
|
|
442
|
+
backslashes++;
|
|
443
|
+
return backslashes % 2 === 0;
|
|
444
|
+
}
|
|
445
|
+
/**
|
|
446
|
+
* Markup Obsidian parses as a node of its own, so a `#` right after it opens a
|
|
447
|
+
* tag the way one does after whitespace, while the same character as plain
|
|
448
|
+
* text blocks the tag. Protecting each span makes the `#` after it a segment
|
|
449
|
+
* start. Pinned against Obsidian 1.13.7 readback; issue #138.
|
|
450
|
+
*
|
|
451
|
+
* A **backslash escape** of ASCII punctuation: `\]#t` and `\\#t` are tags,
|
|
452
|
+
* `\#t` is an escaped hash and not one.
|
|
453
|
+
*/
|
|
454
|
+
const ESCAPE = /(?<escape>\\[!-/:-@[-`{-~])/;
|
|
455
|
+
/**
|
|
456
|
+
* An **inline HTML tag**: `<br>#t`, `x <b>#t</b>`, `<a href="u">#t</a>`. Only
|
|
457
|
+
* a tag CommonMark's grammar accepts is one; `x <a b #t> y` is text.
|
|
458
|
+
*/
|
|
459
|
+
const HTML_TAG = new RegExp(`(?:${HTML_TAG_SOURCE})`);
|
|
460
|
+
/**
|
|
461
|
+
* An **emphasis, highlight, or strikethrough delimiter run** directly before
|
|
462
|
+
* `#` that has a partner elsewhere on its line: `**#t**`, `*#t*`, `**x**#t`,
|
|
463
|
+
* `foo*#t*bar`, `==#t==`, `~~#t~~`. An unpartnered run is literal text to
|
|
464
|
+
* Obsidian (`a *#t`, `a ==#t`), and so is a single `~`. The partner test is
|
|
465
|
+
* the same delimiter anywhere else on the line. `_` pairs by rules of its own,
|
|
466
|
+
* in `underscoreMarks`.
|
|
467
|
+
*/
|
|
468
|
+
const EMPHASIS_RUN = /(?<!\*)(?=\*+#)(?:(?<=\*[^\n]*?)|(?=\*+#[^\n]*?\*))\*+|(?<!=)(?===#)(?:(?<===[^\n]*?)|(?===#[^\n]*?==))==|(?<!~)(?=~~#)(?:(?<=~~[^\n]*?)|(?=~~#[^\n]*?~~))~~/;
|
|
469
|
+
/**
|
|
470
|
+
* A **blockquote marker** directly before `#`: `>#t`, `> >#t`. A table cell
|
|
471
|
+
* pipe is the same kind of node; `markupMarks` protects those, since telling a
|
|
472
|
+
* table row from a pipe in prose takes the block structure around it.
|
|
473
|
+
*/
|
|
474
|
+
const QUOTE_MARKER = />(?=#)(?<=(?:^|\n)[ \t]*(?:>[ \t]*)*>)/;
|
|
475
|
+
/**
|
|
476
|
+
* Every span a tag cannot live in, or that a `#` right after opens a tag, in
|
|
477
|
+
* the order `protectedSpans` tries them: the construct that opens first wins,
|
|
478
|
+
* so `$b <!-- c$` is math and `<!-- $b -->` is a comment.
|
|
479
|
+
*/
|
|
480
|
+
const INLINE_SPANS = new RegExp([
|
|
481
|
+
INLINE_CODE,
|
|
482
|
+
WIKILINK,
|
|
483
|
+
IMAGE,
|
|
484
|
+
LINK_TAIL,
|
|
485
|
+
LINK_DEFINITION,
|
|
486
|
+
HTML_COMMENT,
|
|
487
|
+
INLINE_DOUBLE_MATH,
|
|
488
|
+
INLINE_MATH_OPEN,
|
|
489
|
+
ESCAPE,
|
|
490
|
+
HTML_TAG,
|
|
491
|
+
EMPHASIS_RUN,
|
|
492
|
+
QUOTE_MARKER,
|
|
493
|
+
]
|
|
494
|
+
.map((r) => r.source)
|
|
495
|
+
.join('|'), 'g');
|
|
325
496
|
/**
|
|
326
|
-
*
|
|
327
|
-
*
|
|
328
|
-
* (
|
|
497
|
+
* The characters that end an inline tag, as the body of a character class:
|
|
498
|
+
* whitespace, ASCII punctuation other than `_`, `-`, and `/`, and the General
|
|
499
|
+
* (U+2000–U+206F) and Supplemental (U+2E00–U+2E7F) Punctuation blocks. Every
|
|
500
|
+
* other code point — letters and digits in any script, emoji, variation
|
|
501
|
+
* selectors, combining marks, symbols such as `€` or `©` — continues a tag.
|
|
502
|
+
* Read off Obsidian's own metadata cache (1.13.7): `#café`, `#日本語`, and
|
|
503
|
+
* `#✅done` are tags, `#tag—dash` ends at the em dash, and a zero-width joiner
|
|
504
|
+
* ends a tag mid-emoji. Every regex using it carries the `u` flag, so the class
|
|
505
|
+
* matches whole code points rather than surrogate halves.
|
|
329
506
|
*/
|
|
330
|
-
const
|
|
507
|
+
const TAG_STOP = String.raw `\s!-,.:-@\[-\^\x60{-~\u2000-\u206F\u2E00-\u2E7F`;
|
|
508
|
+
/** One code point that continues an inline tag. */
|
|
509
|
+
const TAG_CHAR = `[^${TAG_STOP}]`;
|
|
331
510
|
/**
|
|
332
|
-
* A tag's left boundary: the start of a segment, or
|
|
333
|
-
*
|
|
334
|
-
*
|
|
511
|
+
* A tag's left boundary: the start of a segment, or whitespace (`\s`, which
|
|
512
|
+
* takes in NBSP and U+3000). A punctuation mark that is plain text before `#`
|
|
513
|
+
* blocks a tag in Obsidian — `(#a`, `.#b`, `x—#c`, an unpartnered `a *#d` —
|
|
514
|
+
* and so does the `\` of an escaped `\#`. A segment starts at the note body or
|
|
515
|
+
* right after a protected span, and Obsidian reads a tag glued to either:
|
|
516
|
+
* `[[x]]#t`, `` `c`#t ``, `**#t**`, `<br>#t`, `[#t]`, `_#t_`, and a table
|
|
517
|
+
* cell's `|#t|` are tags.
|
|
335
518
|
*/
|
|
336
|
-
const TAG_LEFT_BOUNDARY =
|
|
519
|
+
const TAG_LEFT_BOUNDARY = String.raw `(^|\s)`;
|
|
520
|
+
/** Obsidian rejects a tag made only of ASCII digits (`#1984`); `#1990s` and `#١٢٣` are tags. */
|
|
521
|
+
const ALL_DIGITS = /^[0-9]+$/;
|
|
522
|
+
/**
|
|
523
|
+
* A run of tags glued end to end, `#a#b`. Obsidian reads each `#` that
|
|
524
|
+
* directly follows a tag as the start of another one — `#tl#tm` is two tags,
|
|
525
|
+
* `#tn/#to` is `tn/` and `to` — but not one after an all-digit run, since that
|
|
526
|
+
* run was never a tag (`#1984#x` holds none).
|
|
527
|
+
*/
|
|
528
|
+
const TAG_CHAIN_SOURCE = `${TAG_LEFT_BOUNDARY}((?:#${TAG_CHAR}+)+)`;
|
|
529
|
+
const TAG_CHAIN = new RegExp(TAG_CHAIN_SOURCE, 'gu');
|
|
530
|
+
/** A chain plus the single horizontal space after it, which a removal may take. */
|
|
531
|
+
const TAG_CHAIN_SPACED = new RegExp(`${TAG_CHAIN_SOURCE}([ \\t]?)`, 'gu');
|
|
532
|
+
/** The tags in a matched chain: each name, up to the first all-digit one. */
|
|
533
|
+
function chainTags(chain) {
|
|
534
|
+
const tags = [];
|
|
535
|
+
for (const name of chain.slice(1).split('#')) {
|
|
536
|
+
if (ALL_DIGITS.test(name))
|
|
537
|
+
break;
|
|
538
|
+
tags.push(name);
|
|
539
|
+
}
|
|
540
|
+
return tags;
|
|
541
|
+
}
|
|
337
542
|
/**
|
|
338
543
|
* Inline `#tag` syntax lives in the body. The frontmatter block is spliced off
|
|
339
544
|
* first and re-attached verbatim, so a `#` inside a YAML scalar is neither read
|
|
@@ -344,10 +549,9 @@ function mutateInlineTags(content, tags, operation, applied, skipped) {
|
|
|
344
549
|
const segments = splitProtectedSegments(body);
|
|
345
550
|
let updatedNonCode = false;
|
|
346
551
|
if (operation === 'add') {
|
|
552
|
+
const present = new Set(inlineTags(segments));
|
|
347
553
|
for (const tag of tags) {
|
|
348
|
-
|
|
349
|
-
const present = segments.some((s) => !s.protected && re.test(s.text));
|
|
350
|
-
if (present) {
|
|
554
|
+
if (present.has(tag)) {
|
|
351
555
|
skipped.add(tag);
|
|
352
556
|
}
|
|
353
557
|
else {
|
|
@@ -380,7 +584,12 @@ function mutateInlineTags(content, tags, operation, applied, skipped) {
|
|
|
380
584
|
}
|
|
381
585
|
else {
|
|
382
586
|
for (const tag of tags) {
|
|
383
|
-
|
|
587
|
+
/** `#1984` is text to Obsidian and to `list`, so a removal leaves it too. */
|
|
588
|
+
if (ALL_DIGITS.test(tag)) {
|
|
589
|
+
skipped.add(tag);
|
|
590
|
+
continue;
|
|
591
|
+
}
|
|
592
|
+
const remove = removeFromChain(tag);
|
|
384
593
|
let found = false;
|
|
385
594
|
for (const s of segments) {
|
|
386
595
|
if (s.protected)
|
|
@@ -392,7 +601,7 @@ function mutateInlineTags(content, tags, operation, applied, skipped) {
|
|
|
392
601
|
* space. Repeat until the segment stops changing — every pass drops at
|
|
393
602
|
* least the tag itself, so this terminates.
|
|
394
603
|
*/
|
|
395
|
-
for (let next = s.text.replace(
|
|
604
|
+
for (let next = s.text.replace(TAG_CHAIN_SPACED, remove); next !== s.text; next = s.text.replace(TAG_CHAIN_SPACED, remove)) {
|
|
396
605
|
s.text = next;
|
|
397
606
|
found = true;
|
|
398
607
|
}
|
|
@@ -428,65 +637,299 @@ function removeAt(full, leading, trailing, offset, whole) {
|
|
|
428
637
|
return trailing;
|
|
429
638
|
}
|
|
430
639
|
/**
|
|
431
|
-
*
|
|
432
|
-
*
|
|
433
|
-
*
|
|
640
|
+
* A `TAG_CHAIN` replacer that drops every tag named `tag` from a chain and
|
|
641
|
+
* keeps the rest. Chains are split at `#` rather than matched against the
|
|
642
|
+
* name, so removing `caf` cannot strip the front off `#café`. A chain that
|
|
643
|
+
* loses all its tags takes one adjacent space with it, per `removeAt`.
|
|
644
|
+
*/
|
|
645
|
+
function removeFromChain(tag) {
|
|
646
|
+
return (full, leading, chain, trailing, offset, whole) => {
|
|
647
|
+
const names = chain.slice(1).split('#');
|
|
648
|
+
const tagged = chainTags(chain).length;
|
|
649
|
+
const kept = names.filter((name, i) => i >= tagged || name !== tag);
|
|
650
|
+
if (kept.length === names.length)
|
|
651
|
+
return full;
|
|
652
|
+
if (kept.length === 0)
|
|
653
|
+
return removeAt(full, leading, trailing, offset, whole);
|
|
654
|
+
return `${leading}#${kept.join('#')}${trailing}`;
|
|
655
|
+
};
|
|
656
|
+
}
|
|
657
|
+
/** Every inline tag in the segments a tag may live in, each once, in order of first appearance. */
|
|
658
|
+
function inlineTags(segments) {
|
|
659
|
+
const tags = new Set();
|
|
660
|
+
for (const seg of segments) {
|
|
661
|
+
if (seg.protected)
|
|
662
|
+
continue;
|
|
663
|
+
for (const m of seg.text.matchAll(TAG_CHAIN)) {
|
|
664
|
+
for (const tag of chainTags(m[2] ?? ''))
|
|
665
|
+
tags.add(tag);
|
|
666
|
+
}
|
|
667
|
+
}
|
|
668
|
+
return [...tags];
|
|
669
|
+
}
|
|
670
|
+
/**
|
|
671
|
+
* Split the body into stretches a tag may live in and stretches it may not.
|
|
672
|
+
* `scanBlocks` finds the block structure — code, HTML blocks, and display math
|
|
673
|
+
* are hidden whole — and each paragraph, heading, and table row is split by
|
|
674
|
+
* `splitInline`. Both the read and the write path run over the result, so
|
|
434
675
|
* `list` and `remove` agree on what counts as a tag.
|
|
435
676
|
*/
|
|
436
677
|
function splitProtectedSegments(content) {
|
|
437
678
|
const segments = [];
|
|
438
679
|
let cursor = 0;
|
|
439
|
-
const
|
|
440
|
-
.
|
|
441
|
-
|
|
680
|
+
const push = (isProtected, text) => {
|
|
681
|
+
if (text.length > 0)
|
|
682
|
+
segments.push({ protected: isProtected, text });
|
|
683
|
+
};
|
|
684
|
+
for (const block of scanBlocks(content)) {
|
|
685
|
+
push(false, content.slice(cursor, block.start));
|
|
686
|
+
const text = content.slice(block.start, block.end);
|
|
687
|
+
if (block.kind === 'hidden')
|
|
688
|
+
push(true, text);
|
|
689
|
+
else
|
|
690
|
+
for (const seg of splitInline(text, block.tableRow))
|
|
691
|
+
push(seg.protected, seg.text);
|
|
692
|
+
cursor = block.end;
|
|
693
|
+
}
|
|
694
|
+
push(false, content.slice(cursor));
|
|
695
|
+
return segments;
|
|
696
|
+
}
|
|
697
|
+
/**
|
|
698
|
+
* The spans of `text` a tag cannot live in, or that a `#` right after opens a
|
|
699
|
+
* tag, in order.
|
|
700
|
+
*/
|
|
701
|
+
function protectedSpans(text) {
|
|
702
|
+
const spans = [];
|
|
703
|
+
const mathCloser = inlineMathCloser(text);
|
|
704
|
+
const codeCloser = codeSpanCloser(text);
|
|
705
|
+
INLINE_SPANS.lastIndex = 0;
|
|
442
706
|
for (;;) {
|
|
443
|
-
const m =
|
|
707
|
+
const m = INLINE_SPANS.exec(text);
|
|
444
708
|
if (!m)
|
|
445
709
|
break;
|
|
446
|
-
|
|
447
|
-
|
|
448
|
-
|
|
710
|
+
let start = m.index;
|
|
711
|
+
let end = start + m[0].length;
|
|
712
|
+
if (m.groups?.code !== undefined) {
|
|
713
|
+
/**
|
|
714
|
+
* A run with no closer leaves its first backtick as text, and the rest of
|
|
715
|
+
* the run is tried as an opener of its own — where CommonMark would make
|
|
716
|
+
* the whole run text, Obsidian reads `` a ``b #x` `` as the code span
|
|
717
|
+
* `` `b #x` ``. Each shorter opener is tried here rather than by
|
|
718
|
+
* re-running the regex, which would rescan the run from every backtick.
|
|
719
|
+
*/
|
|
720
|
+
const runEnd = end;
|
|
721
|
+
let close = codeCloser(start, runEnd - start);
|
|
722
|
+
while (close === undefined && ++start < runEnd)
|
|
723
|
+
close = codeCloser(start, runEnd - start);
|
|
724
|
+
if (close === undefined)
|
|
725
|
+
continue;
|
|
726
|
+
end = close;
|
|
727
|
+
INLINE_SPANS.lastIndex = end;
|
|
728
|
+
}
|
|
729
|
+
else if (m.groups?.inlineMath !== undefined) {
|
|
730
|
+
/**
|
|
731
|
+
* The `$$` constructs are tried before this one and nothing after it
|
|
732
|
+
* opens with `$`, so an opener with no closer leaves this position to
|
|
733
|
+
* plain text and the scan resumes one character on.
|
|
734
|
+
*/
|
|
735
|
+
const close = mathCloser(m.index);
|
|
736
|
+
if (close === undefined)
|
|
737
|
+
continue;
|
|
738
|
+
end = close + 1;
|
|
739
|
+
INLINE_SPANS.lastIndex = end;
|
|
449
740
|
}
|
|
450
|
-
|
|
451
|
-
|
|
741
|
+
const kind = m.groups?.escape !== undefined
|
|
742
|
+
? 'escape'
|
|
743
|
+
: m.groups?.linkTail !== undefined
|
|
744
|
+
? 'linkTail'
|
|
745
|
+
: 'other';
|
|
746
|
+
spans.push({ start, end, kind });
|
|
452
747
|
}
|
|
453
|
-
|
|
454
|
-
|
|
748
|
+
return spans;
|
|
749
|
+
}
|
|
750
|
+
/**
|
|
751
|
+
* Split one paragraph, heading, or table row. Beyond the spans of
|
|
752
|
+
* `protectedSpans`, three kinds of markup become protected one-character (or
|
|
753
|
+
* one-run) segments, since a `#` directly after each opens a tag: the brackets
|
|
754
|
+
* of a bracketed span, a table row's cell pipes, and an underscore run that
|
|
755
|
+
* pairs as emphasis (`underscoreMarks`).
|
|
756
|
+
*/
|
|
757
|
+
function splitInline(text, tableRow) {
|
|
758
|
+
const spans = protectedSpans(text);
|
|
759
|
+
/** `markEnd[i]` is the end of the mark starting at `i`, or 0. */
|
|
760
|
+
const markEnd = new Uint32Array(text.length);
|
|
761
|
+
const linkTexts = markupMarks(text, spans, tableRow, markEnd);
|
|
762
|
+
underscoreMarks(text, spans, linkTexts, markEnd);
|
|
763
|
+
const segments = [];
|
|
764
|
+
let cursor = 0;
|
|
765
|
+
const cut = (start, end) => {
|
|
766
|
+
if (start > cursor)
|
|
767
|
+
segments.push({ protected: false, text: text.slice(cursor, start) });
|
|
768
|
+
segments.push({ protected: true, text: text.slice(start, end) });
|
|
769
|
+
cursor = end;
|
|
770
|
+
};
|
|
771
|
+
let s = 0;
|
|
772
|
+
for (let i = 0; i < text.length;) {
|
|
773
|
+
const span = spans[s];
|
|
774
|
+
if (span?.start === i) {
|
|
775
|
+
cut(i, span.end);
|
|
776
|
+
i = span.end;
|
|
777
|
+
s++;
|
|
778
|
+
}
|
|
779
|
+
else if (markEnd[i]) {
|
|
780
|
+
const end = markEnd[i];
|
|
781
|
+
cut(i, end);
|
|
782
|
+
i = end;
|
|
783
|
+
}
|
|
784
|
+
else
|
|
785
|
+
i++;
|
|
455
786
|
}
|
|
787
|
+
if (cursor < text.length)
|
|
788
|
+
segments.push({ protected: false, text: text.slice(cursor) });
|
|
456
789
|
return segments;
|
|
457
790
|
}
|
|
458
|
-
/**
|
|
459
|
-
|
|
460
|
-
|
|
461
|
-
|
|
791
|
+
/**
|
|
792
|
+
* Mark the brackets of every bracketed span and, in a table row, every cell
|
|
793
|
+
* pipe; return the link texts — the ranges between a `[` and the `LINK_TAIL`
|
|
794
|
+
* that closes it. Pinned against Obsidian readback: `[#t]`, `x [a]#t`, and
|
|
795
|
+
* `[#t](u)` are tags, as is `[#t` closed on the next line of the paragraph,
|
|
796
|
+
* while `x [#t` (never closed), `x []#t` (empty), `[^#t]` (a footnote), and
|
|
797
|
+
* the outer `[` of `[#t [b] c]` (a `[` before the `]`) open none. A `]]`
|
|
798
|
+
* closes a wikilink, not a bracketed span.
|
|
799
|
+
*/
|
|
800
|
+
function markupMarks(text, spans, tableRow, markEnd) {
|
|
801
|
+
const linkTexts = [];
|
|
802
|
+
let open = -1;
|
|
803
|
+
let s = 0;
|
|
804
|
+
for (let i = 0; i < text.length;) {
|
|
805
|
+
const span = spans[s];
|
|
806
|
+
if (span?.start === i) {
|
|
807
|
+
if (span.kind === 'linkTail') {
|
|
808
|
+
if (open >= 0 && i > open + 1) {
|
|
809
|
+
markEnd[open] = open + 1;
|
|
810
|
+
linkTexts.push([open + 1, i]);
|
|
811
|
+
}
|
|
812
|
+
open = -1;
|
|
813
|
+
}
|
|
814
|
+
i = span.end;
|
|
815
|
+
s++;
|
|
816
|
+
continue;
|
|
817
|
+
}
|
|
818
|
+
const ch = text[i];
|
|
819
|
+
if (ch === '[') {
|
|
820
|
+
open = text[i + 1] === '^' ? -1 : i;
|
|
821
|
+
}
|
|
822
|
+
else if (ch === ']') {
|
|
823
|
+
if (open >= 0 && i > open + 1 && text[i + 1] !== ']') {
|
|
824
|
+
markEnd[open] = open + 1;
|
|
825
|
+
markEnd[i] = i + 1;
|
|
826
|
+
}
|
|
827
|
+
open = -1;
|
|
828
|
+
}
|
|
829
|
+
else if (ch === '|' && tableRow) {
|
|
830
|
+
markEnd[i] = i + 1;
|
|
831
|
+
}
|
|
832
|
+
i++;
|
|
833
|
+
}
|
|
834
|
+
return linkTexts;
|
|
835
|
+
}
|
|
836
|
+
const TAG_AT = new RegExp(`#${TAG_CHAR}+`, 'uy');
|
|
837
|
+
/**
|
|
838
|
+
* Mark the underscore runs that pair as emphasis. `_` is a tag character, so
|
|
839
|
+
* this decides both where a tag may start (`x _#t_ y` is tagged `t`,
|
|
840
|
+
* `snake_case_#t` is tagged `t`) and where one ends (the closing `_` of
|
|
841
|
+
* `_#t_` is not part of it). None of this is CommonMark's flanking rule; it is
|
|
842
|
+
* fitted to more than eighty Obsidian 1.13.7 readback probes. Walking the
|
|
843
|
+
* paragraph's runs in order, with at most one opener pending:
|
|
844
|
+
*
|
|
845
|
+
* - With none pending, a run becomes the opener.
|
|
846
|
+
* - A run followed by an ASCII letter or digit cannot close (`x _#t_b` holds
|
|
847
|
+
* no tag) and is passed over.
|
|
848
|
+
* - A run longer than the opener cannot close it and is passed over.
|
|
849
|
+
* - When the text between opener and run both starts and ends with whitespace
|
|
850
|
+
* (`a_ _#t`), the pair fails and the run becomes the opener instead.
|
|
851
|
+
* - Otherwise the two pair, and both are marks.
|
|
852
|
+
*
|
|
853
|
+
* A run inside a tag's name, or inside a protected span — code, math, a
|
|
854
|
+
* comment, a link destination — can close but never open: `#qb_name_ x` is the
|
|
855
|
+
* tag `qb_name_`, and `x `a_` b_#t` holds no tag while `x _a `_` b_#t` pairs
|
|
856
|
+
* the first two runs. An escaped `\_` is text. Link text pairs on its own, as
|
|
857
|
+
* CommonMark nests it: `[x _a_#t](u)` is tagged `t`, while the run in
|
|
858
|
+
* `[a_](u) b_#t` cannot reach past the link.
|
|
859
|
+
*/
|
|
860
|
+
function underscoreMarks(text, spans, linkTexts, markEnd) {
|
|
861
|
+
let outer;
|
|
862
|
+
let inner;
|
|
863
|
+
/** Index of the link text the walk is in, or -1 outside every link. */
|
|
864
|
+
let link = -1;
|
|
865
|
+
let tagEnd = -1;
|
|
866
|
+
let s = 0;
|
|
867
|
+
let l = 0;
|
|
868
|
+
/** Offer `run` to the pending opener; returns the opener pending afterwards. */
|
|
869
|
+
const offer = (opener, run, closeOnly) => {
|
|
870
|
+
if (!opener)
|
|
871
|
+
return closeOnly ? undefined : run;
|
|
872
|
+
if (/[A-Za-z0-9]/.test(text[run.end] ?? ''))
|
|
873
|
+
return opener;
|
|
874
|
+
if (run.end - run.start > opener.end - opener.start)
|
|
875
|
+
return opener;
|
|
876
|
+
if (/\s/.test(text[opener.end] ?? '') && /\s/.test(text[run.start - 1] ?? '')) {
|
|
877
|
+
return closeOnly ? undefined : run;
|
|
878
|
+
}
|
|
879
|
+
markEnd[opener.start] = opener.end;
|
|
880
|
+
markEnd[run.start] = run.end;
|
|
881
|
+
return;
|
|
882
|
+
};
|
|
883
|
+
for (let i = 0; i < text.length;) {
|
|
884
|
+
while ((spans[s]?.end ?? Number.POSITIVE_INFINITY) <= i)
|
|
885
|
+
s++;
|
|
886
|
+
const span = spans[s];
|
|
887
|
+
const inSpan = span !== undefined && span.start <= i;
|
|
888
|
+
if (inSpan && span.kind === 'escape') {
|
|
889
|
+
i = span.end;
|
|
890
|
+
continue;
|
|
891
|
+
}
|
|
892
|
+
const ch = text[i];
|
|
893
|
+
if (ch === '#' && !inSpan) {
|
|
894
|
+
TAG_AT.lastIndex = i;
|
|
895
|
+
if (TAG_AT.test(text))
|
|
896
|
+
tagEnd = TAG_AT.lastIndex;
|
|
897
|
+
}
|
|
898
|
+
if (ch !== '_') {
|
|
899
|
+
i++;
|
|
900
|
+
continue;
|
|
901
|
+
}
|
|
902
|
+
const run = { start: i, end: i };
|
|
903
|
+
while (text[run.end] === '_' && (!inSpan || run.end < span.end))
|
|
904
|
+
run.end++;
|
|
905
|
+
i = run.end;
|
|
906
|
+
const closeOnly = inSpan || run.start < tagEnd;
|
|
907
|
+
while ((linkTexts[l]?.[1] ?? Number.POSITIVE_INFINITY) <= run.start)
|
|
908
|
+
l++;
|
|
909
|
+
if ((linkTexts[l]?.[0] ?? Number.POSITIVE_INFINITY) <= run.start) {
|
|
910
|
+
if (link !== l)
|
|
911
|
+
inner = undefined;
|
|
912
|
+
link = l;
|
|
913
|
+
inner = offer(inner, run, closeOnly);
|
|
914
|
+
}
|
|
915
|
+
else {
|
|
916
|
+
outer = offer(outer, run, closeOnly);
|
|
917
|
+
}
|
|
918
|
+
}
|
|
462
919
|
}
|
|
463
920
|
/**
|
|
464
921
|
* Read-only helpers for `obsidian_manage_tags list`. Inline tags are read from
|
|
465
922
|
* the body only, and through the same protected-segment split and left boundary
|
|
466
923
|
* a removal uses, so what `list` reports is exactly what `remove` can reach — a
|
|
467
|
-
* `#` inside a YAML scalar,
|
|
468
|
-
*
|
|
924
|
+
* `#` inside a YAML scalar, code, an HTML block or comment, math, an image, or
|
|
925
|
+
* a link's destination or label, or one glued to the character before it
|
|
926
|
+
* (`\#`, `(#x`), is none of them. A tag runs for as long as `TAG_CHAR`
|
|
927
|
+
* matches, and an all-digit run is not a tag.
|
|
469
928
|
*/
|
|
470
929
|
export function listTagsFromContent(content, frontmatter) {
|
|
471
|
-
|
|
472
|
-
|
|
473
|
-
|
|
474
|
-
|
|
475
|
-
const re = new RegExp(`${TAG_LEFT_BOUNDARY}#([a-zA-Z][\\w/-]*)`, 'g');
|
|
476
|
-
for (const seg of splitProtectedSegments(splice(content).body)) {
|
|
477
|
-
if (seg.protected)
|
|
478
|
-
continue;
|
|
479
|
-
for (;;) {
|
|
480
|
-
const m = re.exec(seg.text);
|
|
481
|
-
if (!m)
|
|
482
|
-
break;
|
|
483
|
-
const t = m[2];
|
|
484
|
-
if (t && !seen.has(t)) {
|
|
485
|
-
seen.add(t);
|
|
486
|
-
inline.push(t);
|
|
487
|
-
}
|
|
488
|
-
}
|
|
489
|
-
}
|
|
490
|
-
return { frontmatter: fmTags, inline };
|
|
930
|
+
return {
|
|
931
|
+
frontmatter: normalizeTagList(frontmatter.tags),
|
|
932
|
+
inline: inlineTags(splitProtectedSegments(splice(content).body)),
|
|
933
|
+
};
|
|
491
934
|
}
|
|
492
935
|
//# sourceMappingURL=frontmatter-ops.js.map
|