obsidian-mcp-server 3.5.4 → 3.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. package/AGENTS.md +9 -3
  2. package/CLAUDE.md +9 -3
  3. package/README.md +14 -13
  4. package/changelog/3.5.x/3.5.5.md +20 -0
  5. package/changelog/3.6.x/3.6.0.md +31 -0
  6. package/dist/mcp-server/tools/definitions/_shared/schemas.d.ts.map +1 -1
  7. package/dist/mcp-server/tools/definitions/_shared/schemas.js +2 -2
  8. package/dist/mcp-server/tools/definitions/_shared/schemas.js.map +1 -1
  9. package/dist/mcp-server/tools/definitions/index.d.ts +97 -61
  10. package/dist/mcp-server/tools/definitions/index.d.ts.map +1 -1
  11. package/dist/mcp-server/tools/definitions/obsidian-append-to-note.tool.d.ts +14 -2
  12. package/dist/mcp-server/tools/definitions/obsidian-append-to-note.tool.d.ts.map +1 -1
  13. package/dist/mcp-server/tools/definitions/obsidian-append-to-note.tool.js +17 -4
  14. package/dist/mcp-server/tools/definitions/obsidian-append-to-note.tool.js.map +1 -1
  15. package/dist/mcp-server/tools/definitions/obsidian-get-note.tool.d.ts.map +1 -1
  16. package/dist/mcp-server/tools/definitions/obsidian-get-note.tool.js +4 -3
  17. package/dist/mcp-server/tools/definitions/obsidian-get-note.tool.js.map +1 -1
  18. package/dist/mcp-server/tools/definitions/obsidian-manage-tags.tool.d.ts +5 -2
  19. package/dist/mcp-server/tools/definitions/obsidian-manage-tags.tool.d.ts.map +1 -1
  20. package/dist/mcp-server/tools/definitions/obsidian-manage-tags.tool.js +6 -3
  21. package/dist/mcp-server/tools/definitions/obsidian-manage-tags.tool.js.map +1 -1
  22. package/dist/mcp-server/tools/definitions/obsidian-patch-note.tool.d.ts +17 -4
  23. package/dist/mcp-server/tools/definitions/obsidian-patch-note.tool.d.ts.map +1 -1
  24. package/dist/mcp-server/tools/definitions/obsidian-patch-note.tool.js +20 -6
  25. package/dist/mcp-server/tools/definitions/obsidian-patch-note.tool.js.map +1 -1
  26. package/dist/mcp-server/tools/definitions/obsidian-search-notes.tool.d.ts +6 -5
  27. package/dist/mcp-server/tools/definitions/obsidian-search-notes.tool.d.ts.map +1 -1
  28. package/dist/mcp-server/tools/definitions/obsidian-search-notes.tool.js +26 -24
  29. package/dist/mcp-server/tools/definitions/obsidian-search-notes.tool.js.map +1 -1
  30. package/dist/mcp-server/tools/definitions/obsidian-write-note.tool.d.ts +14 -2
  31. package/dist/mcp-server/tools/definitions/obsidian-write-note.tool.d.ts.map +1 -1
  32. package/dist/mcp-server/tools/definitions/obsidian-write-note.tool.js +26 -13
  33. package/dist/mcp-server/tools/definitions/obsidian-write-note.tool.js.map +1 -1
  34. package/dist/services/obsidian/frontmatter-ops.d.ts +7 -4
  35. package/dist/services/obsidian/frontmatter-ops.d.ts.map +1 -1
  36. package/dist/services/obsidian/frontmatter-ops.js +502 -59
  37. package/dist/services/obsidian/frontmatter-ops.js.map +1 -1
  38. package/dist/services/obsidian/markdown-blocks.d.ts +39 -0
  39. package/dist/services/obsidian/markdown-blocks.d.ts.map +1 -0
  40. package/dist/services/obsidian/markdown-blocks.js +611 -0
  41. package/dist/services/obsidian/markdown-blocks.js.map +1 -0
  42. package/dist/services/obsidian/obsidian-service.d.ts +18 -12
  43. package/dist/services/obsidian/obsidian-service.d.ts.map +1 -1
  44. package/dist/services/obsidian/obsidian-service.js +653 -130
  45. package/dist/services/obsidian/obsidian-service.js.map +1 -1
  46. package/dist/services/obsidian/patch-instruction.d.ts +141 -0
  47. package/dist/services/obsidian/patch-instruction.d.ts.map +1 -0
  48. package/dist/services/obsidian/patch-instruction.js +217 -0
  49. package/dist/services/obsidian/patch-instruction.js.map +1 -0
  50. package/dist/services/obsidian/section-extractor.d.ts +109 -6
  51. package/dist/services/obsidian/section-extractor.d.ts.map +1 -1
  52. package/dist/services/obsidian/section-extractor.js +364 -87
  53. package/dist/services/obsidian/section-extractor.js.map +1 -1
  54. package/dist/services/obsidian/types.d.ts +23 -9
  55. package/dist/services/obsidian/types.d.ts.map +1 -1
  56. package/manifest.json +1 -1
  57. package/package.json +3 -2
  58. package/server.json +3 -3
@@ -5,6 +5,7 @@
5
5
  * @module services/obsidian/frontmatter-ops
6
6
  */
7
7
  import { isMap, isScalar, isSeq, parseDocument, Scalar } from 'yaml';
8
+ import { HTML_TAG_SOURCE, scanBlocks } from './markdown-blocks.js';
8
9
  /**
9
10
  * The one frontmatter boundary: an opening `---` alone on the first line, YAML,
10
11
  * then a `---` that starts its own line. The YAML span and the newline that
@@ -164,8 +165,9 @@ export function deleteFrontmatterKey(content, key) {
164
165
  }
165
166
  /**
166
167
  * Add or remove tags across frontmatter (`tags:` array) and inline `#tag`
167
- * syntax. Inline detection skips code spans, link spans, and a hash escaped as
168
- * `\#` — see `splitProtectedSegments` and `TAG_LEFT_BOUNDARY`.
168
+ * syntax. Inline detection skips code spans, link spans, HTML comments, math,
169
+ * and a `#` preceded by anything but whitespace, line start, or markup such
170
+ * as `**` or `<br>` — see `splitProtectedSegments` and `TAG_LEFT_BOUNDARY`.
169
171
  */
170
172
  export function reconcileTags(content, tags, operation, location) {
171
173
  const norm = (t) => t.replace(/^#+/, '').trim();
@@ -308,8 +310,47 @@ function normalizeTagList(value) {
308
310
  }
309
311
  return [];
310
312
  }
311
- const FENCED_CODE_BLOCK = /```[\s\S]*?```|~~~[\s\S]*?~~~/;
312
- const INLINE_CODE = /`[^`\n]+`/;
313
+ /**
314
+ * The opening backtick run of a code span. It closes at the next whole
315
+ * backtick run of the same length in the paragraph, across line breaks:
316
+ * `` a ``b ` #x`` `` is code, `` a `` #x ``` `` is not. Only the opener is a
317
+ * regex; `codeSpanCloser` finds the closer.
318
+ */
319
+ const INLINE_CODE = /(?<code>`+)/;
320
+ /**
321
+ * The end of the code span each opener in `text` starts, or `undefined` when
322
+ * it has none. Whole backtick runs are listed once, by length; openers must be
323
+ * looked up in ascending order, so each length's list is walked once.
324
+ */
325
+ function codeSpanCloser(text) {
326
+ const runs = new Map();
327
+ for (let i = 0; i < text.length;) {
328
+ if (text[i] !== '`') {
329
+ i++;
330
+ continue;
331
+ }
332
+ const start = i;
333
+ while (text[i] === '`')
334
+ i++;
335
+ const list = runs.get(i - start);
336
+ if (list)
337
+ list.push(start);
338
+ else
339
+ runs.set(i - start, [start]);
340
+ }
341
+ const walked = new Map();
342
+ return (open, length) => {
343
+ const list = runs.get(length);
344
+ if (!list)
345
+ return;
346
+ let k = walked.get(length) ?? 0;
347
+ while ((list[k] ?? Number.POSITIVE_INFINITY) <= open)
348
+ k++;
349
+ walked.set(length, k);
350
+ const close = list[k];
351
+ return close === undefined ? undefined : close + length;
352
+ };
353
+ }
313
354
  /**
314
355
  * `[[Target#Heading|Alias]]`. The `#` in a wikilink opens a heading anchor and
315
356
  * the text after `|` is display text — neither is a tag. Protecting the whole
@@ -320,20 +361,184 @@ const INLINE_CODE = /`[^`\n]+`/;
320
361
  * link target, so a span never nests.
321
362
  */
322
363
  const WIKILINK = /\[\[[^[\]\n]*\]\]/;
323
- /** `[text](destination)` — a `#` in the link text or in a URL fragment is link syntax. */
324
- const MARKDOWN_LINK = /\[[^[\]\n]*\]\([^()\n]*\)/;
364
+ /** `![alt](source)` and `![alt][label]` — Obsidian reads no tag in an image's alt text. */
365
+ const IMAGE = /!\[[^[\]\n]*\](?:\([^()\n]*\)|\[[^[\]\n]*\])/;
366
+ /**
367
+ * The `](destination)` or `][label]` that makes the bracketed text before it a
368
+ * link. Only this tail is link syntax: Obsidian reads a tag in link text
369
+ * (`[Discord #tf](https://x.y)` is tagged `tf`) and none in a destination, a
370
+ * title, or a label. The text must open on the same line with no bracket in it;
371
+ * a bracketed phrase followed by a space and a second one (`[note #work] [ref]`)
372
+ * is no link.
373
+ */
374
+ const LINK_TAIL = /(?<linkTail>\](?<=\[[^[\]\n]*\])(?:\([^()\n]*\)|\[[^[\]\n]*\]))/;
375
+ /** The `[label]:` that opens a link reference definition. */
376
+ const LINK_DEFINITION = /\[(?<=(?:^|\n)[ ]{0,3}\[)[^[\]\n]+\]:/;
377
+ /**
378
+ * Obsidian's metadata cache reads no tag inside an HTML comment or math. The
379
+ * rules below are pinned against its readback (Obsidian 1.13.7) rather than a
380
+ * spec. HTML blocks and display math blocks are block structure and live in
381
+ * `markdown-blocks.ts`; these are the spans inside one paragraph. Issue #138.
382
+ *
383
+ * An **inline comment**: `<!--`, then text that does not open with `>` or `->`
384
+ * and holds no `--`, then `-->`. `<!-- x -- y -->` is not a comment, and
385
+ * neither is an unclosed `<!--`.
386
+ */
387
+ const HTML_COMMENT = /<!--(?!-?>)(?:(?!--|\n *\r?\n)[\s\S])*-->/;
388
+ /** Unescaped `$$ … $$` within one paragraph, with no spacing rule. */
389
+ const INLINE_DOUBLE_MATH = /(?<=(?:^|[^\\])(?:\\\\)*)\$\$(?:(?!\n *\r?\n)[\s\S])*?\$\$/;
390
+ /**
391
+ * Unescaped `$ … $` within one paragraph. The opener is not followed by a space
392
+ * or tab; the closer is not preceded by one and not followed by a digit, and a
393
+ * `$` that fails those is passed over rather than ending the span. That is what
394
+ * keeps `cost $5 and #rc for $10` out of math while `m $a #ra b$ n` is in it.
395
+ *
396
+ * Only the opener is a regex; `inlineMathCloser` finds the closer. A lazy
397
+ * regex body would rescan to the end of the paragraph from every opener that
398
+ * never closes — quadratic in a table of prices (issue #143).
399
+ */
400
+ const INLINE_MATH_OPEN = /(?<inlineMath>(?<=(?:^|[^\\])(?:\\\\)*)\$(?![ \t$]))/;
401
+ /** The start of an empty or spaces-only line, which inline math cannot cross. */
402
+ const PARAGRAPH_BREAK = /\n *\r?\n/y;
403
+ /**
404
+ * The closer of the inline math span each opener in `content` starts, or
405
+ * `undefined` when it has none. Whether a `$` can close a span does not depend
406
+ * on where the span opened, so the closers and paragraph breaks are listed
407
+ * once and each opener takes the first closer after it, provided no break
408
+ * comes first. Openers must be looked up in ascending order.
409
+ */
410
+ function inlineMathCloser(content) {
411
+ const closers = [];
412
+ const breaks = [];
413
+ for (let i = 0; i < content.length; i++) {
414
+ if (content[i] === '$' && closesInlineMath(content, i))
415
+ closers.push(i);
416
+ if (content[i] === '\n') {
417
+ PARAGRAPH_BREAK.lastIndex = i;
418
+ if (PARAGRAPH_BREAK.test(content))
419
+ breaks.push(i);
420
+ }
421
+ }
422
+ let c = 0;
423
+ let b = 0;
424
+ return (open) => {
425
+ while ((closers[c] ?? Infinity) <= open)
426
+ c++;
427
+ while ((breaks[b] ?? Infinity) <= open)
428
+ b++;
429
+ const close = closers[c];
430
+ if (close === undefined || (breaks[b] ?? Infinity) < close)
431
+ return;
432
+ return close;
433
+ };
434
+ }
435
+ /** Whether the `$` at `i` can close inline math: unescaped, after no space or tab, before no digit. */
436
+ function closesInlineMath(content, i) {
437
+ const before = content[i - 1];
438
+ if (before === ' ' || before === '\t' || /[0-9]/.test(content[i + 1] ?? ''))
439
+ return false;
440
+ let backslashes = 0;
441
+ while (content[i - 1 - backslashes] === '\\')
442
+ backslashes++;
443
+ return backslashes % 2 === 0;
444
+ }
445
+ /**
446
+ * Markup Obsidian parses as a node of its own, so a `#` right after it opens a
447
+ * tag the way one does after whitespace, while the same character as plain
448
+ * text blocks the tag. Protecting each span makes the `#` after it a segment
449
+ * start. Pinned against Obsidian 1.13.7 readback; issue #138.
450
+ *
451
+ * A **backslash escape** of ASCII punctuation: `\]#t` and `\\#t` are tags,
452
+ * `\#t` is an escaped hash and not one.
453
+ */
454
+ const ESCAPE = /(?<escape>\\[!-/:-@[-`{-~])/;
455
+ /**
456
+ * An **inline HTML tag**: `<br>#t`, `x <b>#t</b>`, `<a href="u">#t</a>`. Only
457
+ * a tag CommonMark's grammar accepts is one; `x <a b #t> y` is text.
458
+ */
459
+ const HTML_TAG = new RegExp(`(?:${HTML_TAG_SOURCE})`);
460
+ /**
461
+ * An **emphasis, highlight, or strikethrough delimiter run** directly before
462
+ * `#` that has a partner elsewhere on its line: `**#t**`, `*#t*`, `**x**#t`,
463
+ * `foo*#t*bar`, `==#t==`, `~~#t~~`. An unpartnered run is literal text to
464
+ * Obsidian (`a *#t`, `a ==#t`), and so is a single `~`. The partner test is
465
+ * the same delimiter anywhere else on the line. `_` pairs by rules of its own,
466
+ * in `underscoreMarks`.
467
+ */
468
+ const EMPHASIS_RUN = /(?<!\*)(?=\*+#)(?:(?<=\*[^\n]*?)|(?=\*+#[^\n]*?\*))\*+|(?<!=)(?===#)(?:(?<===[^\n]*?)|(?===#[^\n]*?==))==|(?<!~)(?=~~#)(?:(?<=~~[^\n]*?)|(?=~~#[^\n]*?~~))~~/;
469
+ /**
470
+ * A **blockquote marker** directly before `#`: `>#t`, `> >#t`. A table cell
471
+ * pipe is the same kind of node; `markupMarks` protects those, since telling a
472
+ * table row from a pipe in prose takes the block structure around it.
473
+ */
474
+ const QUOTE_MARKER = />(?=#)(?<=(?:^|\n)[ \t]*(?:>[ \t]*)*>)/;
475
+ /**
476
+ * Every span a tag cannot live in, or that a `#` right after opens a tag, in
477
+ * the order `protectedSpans` tries them: the construct that opens first wins,
478
+ * so `$b <!-- c$` is math and `<!-- $b -->` is a comment.
479
+ */
480
+ const INLINE_SPANS = new RegExp([
481
+ INLINE_CODE,
482
+ WIKILINK,
483
+ IMAGE,
484
+ LINK_TAIL,
485
+ LINK_DEFINITION,
486
+ HTML_COMMENT,
487
+ INLINE_DOUBLE_MATH,
488
+ INLINE_MATH_OPEN,
489
+ ESCAPE,
490
+ HTML_TAG,
491
+ EMPHASIS_RUN,
492
+ QUOTE_MARKER,
493
+ ]
494
+ .map((r) => r.source)
495
+ .join('|'), 'g');
325
496
  /**
326
- * `[text][label]` and `[text][]` — the reference-style forms of the same link.
327
- * The label must follow the text immediately; a bracketed phrase on its own
328
- * (`[note #work]`) is not a link and its `#` stays a tag.
497
+ * The characters that end an inline tag, as the body of a character class:
498
+ * whitespace, ASCII punctuation other than `_`, `-`, and `/`, and the General
499
+ * (U+2000–U+206F) and Supplemental (U+2E00–U+2E7F) Punctuation blocks. Every
500
+ * other code point — letters and digits in any script, emoji, variation
501
+ * selectors, combining marks, symbols such as `€` or `©` — continues a tag.
502
+ * Read off Obsidian's own metadata cache (1.13.7): `#café`, `#日本語`, and
503
+ * `#✅done` are tags, `#tag—dash` ends at the em dash, and a zero-width joiner
504
+ * ends a tag mid-emoji. Every regex using it carries the `u` flag, so the class
505
+ * matches whole code points rather than surrogate halves.
329
506
  */
330
- const REFERENCE_LINK = /\[[^[\]\n]+\]\[[^[\]\n]*\]/;
507
+ const TAG_STOP = String.raw `\s!-,.:-@\[-\^\x60{-~\u2000-\u206F\u2E00-\u2E7F`;
508
+ /** One code point that continues an inline tag. */
509
+ const TAG_CHAR = `[^${TAG_STOP}]`;
331
510
  /**
332
- * A tag's left boundary: the start of a segment, or a character that is neither
333
- * part of a tag nor the `\` that escapes one. Obsidian documents `\#` as an
334
- * escaped hashtag — a literal `#` carrying no formatting.
511
+ * A tag's left boundary: the start of a segment, or whitespace (`\s`, which
512
+ * takes in NBSP and U+3000). A punctuation mark that is plain text before `#`
513
+ * blocks a tag in Obsidian — `(#a`, `.#b`, `x—#c`, an unpartnered `a *#d` —
514
+ * and so does the `\` of an escaped `\#`. A segment starts at the note body or
515
+ * right after a protected span, and Obsidian reads a tag glued to either:
516
+ * `[[x]]#t`, `` `c`#t ``, `**#t**`, `<br>#t`, `[#t]`, `_#t_`, and a table
517
+ * cell's `|#t|` are tags.
335
518
  */
336
- const TAG_LEFT_BOUNDARY = '(^|[^\\w/\\\\])';
519
+ const TAG_LEFT_BOUNDARY = String.raw `(^|\s)`;
520
+ /** Obsidian rejects a tag made only of ASCII digits (`#1984`); `#1990s` and `#١٢٣` are tags. */
521
+ const ALL_DIGITS = /^[0-9]+$/;
522
+ /**
523
+ * A run of tags glued end to end, `#a#b`. Obsidian reads each `#` that
524
+ * directly follows a tag as the start of another one — `#tl#tm` is two tags,
525
+ * `#tn/#to` is `tn/` and `to` — but not one after an all-digit run, since that
526
+ * run was never a tag (`#1984#x` holds none).
527
+ */
528
+ const TAG_CHAIN_SOURCE = `${TAG_LEFT_BOUNDARY}((?:#${TAG_CHAR}+)+)`;
529
+ const TAG_CHAIN = new RegExp(TAG_CHAIN_SOURCE, 'gu');
530
+ /** A chain plus the single horizontal space after it, which a removal may take. */
531
+ const TAG_CHAIN_SPACED = new RegExp(`${TAG_CHAIN_SOURCE}([ \\t]?)`, 'gu');
532
+ /** The tags in a matched chain: each name, up to the first all-digit one. */
533
+ function chainTags(chain) {
534
+ const tags = [];
535
+ for (const name of chain.slice(1).split('#')) {
536
+ if (ALL_DIGITS.test(name))
537
+ break;
538
+ tags.push(name);
539
+ }
540
+ return tags;
541
+ }
337
542
  /**
338
543
  * Inline `#tag` syntax lives in the body. The frontmatter block is spliced off
339
544
  * first and re-attached verbatim, so a `#` inside a YAML scalar is neither read
@@ -344,10 +549,9 @@ function mutateInlineTags(content, tags, operation, applied, skipped) {
344
549
  const segments = splitProtectedSegments(body);
345
550
  let updatedNonCode = false;
346
551
  if (operation === 'add') {
552
+ const present = new Set(inlineTags(segments));
347
553
  for (const tag of tags) {
348
- const re = makeInlineTagRegex(tag);
349
- const present = segments.some((s) => !s.protected && re.test(s.text));
350
- if (present) {
554
+ if (present.has(tag)) {
351
555
  skipped.add(tag);
352
556
  }
353
557
  else {
@@ -380,7 +584,12 @@ function mutateInlineTags(content, tags, operation, applied, skipped) {
380
584
  }
381
585
  else {
382
586
  for (const tag of tags) {
383
- const re = makeInlineTagRegex(tag);
587
+ /** `#1984` is text to Obsidian and to `list`, so a removal leaves it too. */
588
+ if (ALL_DIGITS.test(tag)) {
589
+ skipped.add(tag);
590
+ continue;
591
+ }
592
+ const remove = removeFromChain(tag);
384
593
  let found = false;
385
594
  for (const s of segments) {
386
595
  if (s.protected)
@@ -392,7 +601,7 @@ function mutateInlineTags(content, tags, operation, applied, skipped) {
392
601
  * space. Repeat until the segment stops changing — every pass drops at
393
602
  * least the tag itself, so this terminates.
394
603
  */
395
- for (let next = s.text.replace(re, removeAt); next !== s.text; next = s.text.replace(re, removeAt)) {
604
+ for (let next = s.text.replace(TAG_CHAIN_SPACED, remove); next !== s.text; next = s.text.replace(TAG_CHAIN_SPACED, remove)) {
396
605
  s.text = next;
397
606
  found = true;
398
607
  }
@@ -428,65 +637,299 @@ function removeAt(full, leading, trailing, offset, whole) {
428
637
  return trailing;
429
638
  }
430
639
  /**
431
- * Split the body into stretches a tag may live in and stretches it may not:
432
- * code, where a `#` is code, and link syntax, where a `#` is a heading anchor
433
- * or link text. Both the read and the write path run over the result, so
640
+ * A `TAG_CHAIN` replacer that drops every tag named `tag` from a chain and
641
+ * keeps the rest. Chains are split at `#` rather than matched against the
642
+ * name, so removing `caf` cannot strip the front off `#café`. A chain that
643
+ * loses all its tags takes one adjacent space with it, per `removeAt`.
644
+ */
645
+ function removeFromChain(tag) {
646
+ return (full, leading, chain, trailing, offset, whole) => {
647
+ const names = chain.slice(1).split('#');
648
+ const tagged = chainTags(chain).length;
649
+ const kept = names.filter((name, i) => i >= tagged || name !== tag);
650
+ if (kept.length === names.length)
651
+ return full;
652
+ if (kept.length === 0)
653
+ return removeAt(full, leading, trailing, offset, whole);
654
+ return `${leading}#${kept.join('#')}${trailing}`;
655
+ };
656
+ }
657
+ /** Every inline tag in the segments a tag may live in, each once, in order of first appearance. */
658
+ function inlineTags(segments) {
659
+ const tags = new Set();
660
+ for (const seg of segments) {
661
+ if (seg.protected)
662
+ continue;
663
+ for (const m of seg.text.matchAll(TAG_CHAIN)) {
664
+ for (const tag of chainTags(m[2] ?? ''))
665
+ tags.add(tag);
666
+ }
667
+ }
668
+ return [...tags];
669
+ }
670
+ /**
671
+ * Split the body into stretches a tag may live in and stretches it may not.
672
+ * `scanBlocks` finds the block structure — code, HTML blocks, and display math
673
+ * are hidden whole — and each paragraph, heading, and table row is split by
674
+ * `splitInline`. Both the read and the write path run over the result, so
434
675
  * `list` and `remove` agree on what counts as a tag.
435
676
  */
436
677
  function splitProtectedSegments(content) {
437
678
  const segments = [];
438
679
  let cursor = 0;
439
- const re = new RegExp([FENCED_CODE_BLOCK, INLINE_CODE, WIKILINK, MARKDOWN_LINK, REFERENCE_LINK]
440
- .map((r) => r.source)
441
- .join('|'), 'g');
680
+ const push = (isProtected, text) => {
681
+ if (text.length > 0)
682
+ segments.push({ protected: isProtected, text });
683
+ };
684
+ for (const block of scanBlocks(content)) {
685
+ push(false, content.slice(cursor, block.start));
686
+ const text = content.slice(block.start, block.end);
687
+ if (block.kind === 'hidden')
688
+ push(true, text);
689
+ else
690
+ for (const seg of splitInline(text, block.tableRow))
691
+ push(seg.protected, seg.text);
692
+ cursor = block.end;
693
+ }
694
+ push(false, content.slice(cursor));
695
+ return segments;
696
+ }
697
+ /**
698
+ * The spans of `text` a tag cannot live in, or that a `#` right after opens a
699
+ * tag, in order.
700
+ */
701
+ function protectedSpans(text) {
702
+ const spans = [];
703
+ const mathCloser = inlineMathCloser(text);
704
+ const codeCloser = codeSpanCloser(text);
705
+ INLINE_SPANS.lastIndex = 0;
442
706
  for (;;) {
443
- const m = re.exec(content);
707
+ const m = INLINE_SPANS.exec(text);
444
708
  if (!m)
445
709
  break;
446
- const matched = m[0] ?? '';
447
- if (m.index > cursor) {
448
- segments.push({ protected: false, text: content.slice(cursor, m.index) });
710
+ let start = m.index;
711
+ let end = start + m[0].length;
712
+ if (m.groups?.code !== undefined) {
713
+ /**
714
+ * A run with no closer leaves its first backtick as text, and the rest of
715
+ * the run is tried as an opener of its own — where CommonMark would make
716
+ * the whole run text, Obsidian reads `` a ``b #x` `` as the code span
717
+ * `` `b #x` ``. Each shorter opener is tried here rather than by
718
+ * re-running the regex, which would rescan the run from every backtick.
719
+ */
720
+ const runEnd = end;
721
+ let close = codeCloser(start, runEnd - start);
722
+ while (close === undefined && ++start < runEnd)
723
+ close = codeCloser(start, runEnd - start);
724
+ if (close === undefined)
725
+ continue;
726
+ end = close;
727
+ INLINE_SPANS.lastIndex = end;
728
+ }
729
+ else if (m.groups?.inlineMath !== undefined) {
730
+ /**
731
+ * The `$$` constructs are tried before this one and nothing after it
732
+ * opens with `$`, so an opener with no closer leaves this position to
733
+ * plain text and the scan resumes one character on.
734
+ */
735
+ const close = mathCloser(m.index);
736
+ if (close === undefined)
737
+ continue;
738
+ end = close + 1;
739
+ INLINE_SPANS.lastIndex = end;
449
740
  }
450
- segments.push({ protected: true, text: matched });
451
- cursor = m.index + matched.length;
741
+ const kind = m.groups?.escape !== undefined
742
+ ? 'escape'
743
+ : m.groups?.linkTail !== undefined
744
+ ? 'linkTail'
745
+ : 'other';
746
+ spans.push({ start, end, kind });
452
747
  }
453
- if (cursor < content.length) {
454
- segments.push({ protected: false, text: content.slice(cursor) });
748
+ return spans;
749
+ }
750
+ /**
751
+ * Split one paragraph, heading, or table row. Beyond the spans of
752
+ * `protectedSpans`, three kinds of markup become protected one-character (or
753
+ * one-run) segments, since a `#` directly after each opens a tag: the brackets
754
+ * of a bracketed span, a table row's cell pipes, and an underscore run that
755
+ * pairs as emphasis (`underscoreMarks`).
756
+ */
757
+ function splitInline(text, tableRow) {
758
+ const spans = protectedSpans(text);
759
+ /** `markEnd[i]` is the end of the mark starting at `i`, or 0. */
760
+ const markEnd = new Uint32Array(text.length);
761
+ const linkTexts = markupMarks(text, spans, tableRow, markEnd);
762
+ underscoreMarks(text, spans, linkTexts, markEnd);
763
+ const segments = [];
764
+ let cursor = 0;
765
+ const cut = (start, end) => {
766
+ if (start > cursor)
767
+ segments.push({ protected: false, text: text.slice(cursor, start) });
768
+ segments.push({ protected: true, text: text.slice(start, end) });
769
+ cursor = end;
770
+ };
771
+ let s = 0;
772
+ for (let i = 0; i < text.length;) {
773
+ const span = spans[s];
774
+ if (span?.start === i) {
775
+ cut(i, span.end);
776
+ i = span.end;
777
+ s++;
778
+ }
779
+ else if (markEnd[i]) {
780
+ const end = markEnd[i];
781
+ cut(i, end);
782
+ i = end;
783
+ }
784
+ else
785
+ i++;
455
786
  }
787
+ if (cursor < text.length)
788
+ segments.push({ protected: false, text: text.slice(cursor) });
456
789
  return segments;
457
790
  }
458
- /** Captures the character before the tag and the single horizontal space after it, if any. */
459
- function makeInlineTagRegex(tag) {
460
- const escaped = tag.replace(/[\\^$.*+?()[\]{}|]/g, '\\$&');
461
- return new RegExp(`${TAG_LEFT_BOUNDARY}#${escaped}(?![\\w/-])([ \\t]?)`, 'g');
791
+ /**
792
+ * Mark the brackets of every bracketed span and, in a table row, every cell
793
+ * pipe; return the link texts — the ranges between a `[` and the `LINK_TAIL`
794
+ * that closes it. Pinned against Obsidian readback: `[#t]`, `x [a]#t`, and
795
+ * `[#t](u)` are tags, as is `[#t` closed on the next line of the paragraph,
796
+ * while `x [#t` (never closed), `x []#t` (empty), `[^#t]` (a footnote), and
797
+ * the outer `[` of `[#t [b] c]` (a `[` before the `]`) open none. A `]]`
798
+ * closes a wikilink, not a bracketed span.
799
+ */
800
+ function markupMarks(text, spans, tableRow, markEnd) {
801
+ const linkTexts = [];
802
+ let open = -1;
803
+ let s = 0;
804
+ for (let i = 0; i < text.length;) {
805
+ const span = spans[s];
806
+ if (span?.start === i) {
807
+ if (span.kind === 'linkTail') {
808
+ if (open >= 0 && i > open + 1) {
809
+ markEnd[open] = open + 1;
810
+ linkTexts.push([open + 1, i]);
811
+ }
812
+ open = -1;
813
+ }
814
+ i = span.end;
815
+ s++;
816
+ continue;
817
+ }
818
+ const ch = text[i];
819
+ if (ch === '[') {
820
+ open = text[i + 1] === '^' ? -1 : i;
821
+ }
822
+ else if (ch === ']') {
823
+ if (open >= 0 && i > open + 1 && text[i + 1] !== ']') {
824
+ markEnd[open] = open + 1;
825
+ markEnd[i] = i + 1;
826
+ }
827
+ open = -1;
828
+ }
829
+ else if (ch === '|' && tableRow) {
830
+ markEnd[i] = i + 1;
831
+ }
832
+ i++;
833
+ }
834
+ return linkTexts;
835
+ }
836
+ const TAG_AT = new RegExp(`#${TAG_CHAR}+`, 'uy');
837
+ /**
838
+ * Mark the underscore runs that pair as emphasis. `_` is a tag character, so
839
+ * this decides both where a tag may start (`x _#t_ y` is tagged `t`,
840
+ * `snake_case_#t` is tagged `t`) and where one ends (the closing `_` of
841
+ * `_#t_` is not part of it). None of this is CommonMark's flanking rule; it is
842
+ * fitted to more than eighty Obsidian 1.13.7 readback probes. Walking the
843
+ * paragraph's runs in order, with at most one opener pending:
844
+ *
845
+ * - With none pending, a run becomes the opener.
846
+ * - A run followed by an ASCII letter or digit cannot close (`x _#t_b` holds
847
+ * no tag) and is passed over.
848
+ * - A run longer than the opener cannot close it and is passed over.
849
+ * - When the text between opener and run both starts and ends with whitespace
850
+ * (`a_ _#t`), the pair fails and the run becomes the opener instead.
851
+ * - Otherwise the two pair, and both are marks.
852
+ *
853
+ * A run inside a tag's name, or inside a protected span — code, math, a
854
+ * comment, a link destination — can close but never open: `#qb_name_ x` is the
855
+ * tag `qb_name_`, and `x `a_` b_#t` holds no tag while `x _a `_` b_#t` pairs
856
+ * the first two runs. An escaped `\_` is text. Link text pairs on its own, as
857
+ * CommonMark nests it: `[x _a_#t](u)` is tagged `t`, while the run in
858
+ * `[a_](u) b_#t` cannot reach past the link.
859
+ */
860
+ function underscoreMarks(text, spans, linkTexts, markEnd) {
861
+ let outer;
862
+ let inner;
863
+ /** Index of the link text the walk is in, or -1 outside every link. */
864
+ let link = -1;
865
+ let tagEnd = -1;
866
+ let s = 0;
867
+ let l = 0;
868
+ /** Offer `run` to the pending opener; returns the opener pending afterwards. */
869
+ const offer = (opener, run, closeOnly) => {
870
+ if (!opener)
871
+ return closeOnly ? undefined : run;
872
+ if (/[A-Za-z0-9]/.test(text[run.end] ?? ''))
873
+ return opener;
874
+ if (run.end - run.start > opener.end - opener.start)
875
+ return opener;
876
+ if (/\s/.test(text[opener.end] ?? '') && /\s/.test(text[run.start - 1] ?? '')) {
877
+ return closeOnly ? undefined : run;
878
+ }
879
+ markEnd[opener.start] = opener.end;
880
+ markEnd[run.start] = run.end;
881
+ return;
882
+ };
883
+ for (let i = 0; i < text.length;) {
884
+ while ((spans[s]?.end ?? Number.POSITIVE_INFINITY) <= i)
885
+ s++;
886
+ const span = spans[s];
887
+ const inSpan = span !== undefined && span.start <= i;
888
+ if (inSpan && span.kind === 'escape') {
889
+ i = span.end;
890
+ continue;
891
+ }
892
+ const ch = text[i];
893
+ if (ch === '#' && !inSpan) {
894
+ TAG_AT.lastIndex = i;
895
+ if (TAG_AT.test(text))
896
+ tagEnd = TAG_AT.lastIndex;
897
+ }
898
+ if (ch !== '_') {
899
+ i++;
900
+ continue;
901
+ }
902
+ const run = { start: i, end: i };
903
+ while (text[run.end] === '_' && (!inSpan || run.end < span.end))
904
+ run.end++;
905
+ i = run.end;
906
+ const closeOnly = inSpan || run.start < tagEnd;
907
+ while ((linkTexts[l]?.[1] ?? Number.POSITIVE_INFINITY) <= run.start)
908
+ l++;
909
+ if ((linkTexts[l]?.[0] ?? Number.POSITIVE_INFINITY) <= run.start) {
910
+ if (link !== l)
911
+ inner = undefined;
912
+ link = l;
913
+ inner = offer(inner, run, closeOnly);
914
+ }
915
+ else {
916
+ outer = offer(outer, run, closeOnly);
917
+ }
918
+ }
462
919
  }
463
920
  /**
464
921
  * Read-only helpers for `obsidian_manage_tags list`. Inline tags are read from
465
922
  * the body only, and through the same protected-segment split and left boundary
466
923
  * a removal uses, so what `list` reports is exactly what `remove` can reach — a
467
- * `#` inside a YAML scalar, a code span, a link span, or escaped as `\#` is
468
- * none of them.
924
+ * `#` inside a YAML scalar, code, an HTML block or comment, math, an image, or
925
+ * a link's destination or label, or one glued to the character before it
926
+ * (`\#`, `(#x`), is none of them. A tag runs for as long as `TAG_CHAR`
927
+ * matches, and an all-digit run is not a tag.
469
928
  */
470
929
  export function listTagsFromContent(content, frontmatter) {
471
- const fmTags = normalizeTagList(frontmatter.tags);
472
- const inline = [];
473
- const seen = new Set();
474
- /** Each segment is scanned to exhaustion, which resets `lastIndex` between them. */
475
- const re = new RegExp(`${TAG_LEFT_BOUNDARY}#([a-zA-Z][\\w/-]*)`, 'g');
476
- for (const seg of splitProtectedSegments(splice(content).body)) {
477
- if (seg.protected)
478
- continue;
479
- for (;;) {
480
- const m = re.exec(seg.text);
481
- if (!m)
482
- break;
483
- const t = m[2];
484
- if (t && !seen.has(t)) {
485
- seen.add(t);
486
- inline.push(t);
487
- }
488
- }
489
- }
490
- return { frontmatter: fmTags, inline };
930
+ return {
931
+ frontmatter: normalizeTagList(frontmatter.tags),
932
+ inline: inlineTags(splitProtectedSegments(splice(content).body)),
933
+ };
491
934
  }
492
935
  //# sourceMappingURL=frontmatter-ops.js.map