@openparachute/vault 0.7.3-rc.10 → 0.7.3-rc.12

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -64,41 +64,57 @@ import {
64
64
  } from "../core/src/transcription/provider.ts";
65
65
  import { ScribeHttpProvider } from "./transcription/providers/scribe-http.ts";
66
66
 
67
- /** Placeholder pattern written by the voice-memo capture stub. */
68
- const TRANSCRIPT_PLACEHOLDER = /_Transcript pending\._/;
69
-
70
67
  /**
71
- * Body written when transcription reaches a terminal failure (maxAttempts
72
- * exhausted, or the audio file is missing). This used to be written by
73
- * Lens's now-removed scribe client; owning it here means a failed upload
74
- * stops reading "Transcript pending" forever regardless of which client
75
- * uploaded the audio.
68
+ * The in-body transcription markers.
69
+ *
70
+ * The BARE markers are the un-segmented default; voice W2 (segmented
71
+ * recordings) targets per-part variants built by `markersFor`. Both are a
72
+ * BYTE-EXACT cross-door + cross-repo contract — the cloud Workers-AI
73
+ * transcription path ships the identical strings, and the notes-ui status
74
+ * chip (parachute-surface TranscriptionStatus.tsx) keys off the failure
75
+ * marker's exact copy. Don't change any of this text without a coordinated
76
+ * change in both places. A friendlier "retry available" copy + chip
77
+ * affordance is a tracked parachute-surface follow-up.
76
78
  *
77
- * NOTE: the notes-ui status chip (parachute-surface TranscriptionStatus.tsx)
78
- * keys off this exact string, so don't change the copy without a coordinated
79
- * change there. A friendlier "retry available" copy + chip affordance is a
80
- * tracked parachute-surface follow-up.
79
+ * Owning the failure marker here (it used to be written by Lens's now-removed
80
+ * scribe client) means a failed upload stops reading "Transcript pending"
81
+ * forever regardless of which client uploaded the audio.
81
82
  */
82
- const TRANSCRIPT_UNAVAILABLE = "_Transcription unavailable._";
83
+ const BARE_PENDING = "_Transcript pending._";
84
+ const BARE_UNAVAILABLE = "_Transcription unavailable._";
85
+
86
+ /** Escape a literal string for safe embedding in a `RegExp`. */
87
+ function escapeRegExp(literal: string): string {
88
+ return literal.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
89
+ }
83
90
 
84
91
  /**
85
- * On a successful (re)transcription of a legacy in-body memo, the transcript
86
- * replaces whichever marker is currently in the body the original
87
- * `_Transcript pending._` on a first-try success, OR `_Transcription
88
- * unavailable._` if a prior attempt failed and we're now retrying. Matching
89
- * both means a retried success lands in the same spot a first-try success
90
- * would, preserving the surrounding capture body (the `![[memo]]` embed,
91
- * the `_Recorded …_` line, the header).
92
- *
93
- * Deliberately NO `/g` flag — `.replace` swaps only the FIRST match. A
94
- * canonical capture body holds exactly one marker, so first-match is the
95
- * correct target. `applyFailureMarker`'s includes-guard (no-op when the
96
- * marker is already present) prevents markers accumulating across repeated
97
- * terminal failures, so the body never carries two of the same marker. A
98
- * hand-edited body that somehow contains both markers patches only the
99
- * first — accepted (degenerate, operator-induced).
92
+ * The pending + terminal-failure markers for an attachment, honoring
93
+ * `segment_index` (voice W2 segmented recordings). `undefined` yields the
94
+ * bare markers (fully backward compatible — un-segmented flows are byte-
95
+ * unchanged). An integer 0 yields this part's markers, with the human-
96
+ * facing part number N = segment_index + 1 (1-based, decimal):
97
+ * `_Transcript pending (part N)._` / `_Transcription unavailable (part N)._`
100
98
  */
101
- const TRANSCRIPT_SUCCESS_TARGET = /_Transcript pending\._|_Transcription unavailable\._/;
99
+ function markersFor(segmentIndex: number | undefined): { pending: string; unavailable: string } {
100
+ if (segmentIndex === undefined) return { pending: BARE_PENDING, unavailable: BARE_UNAVAILABLE };
101
+ const n = segmentIndex + 1;
102
+ return {
103
+ pending: `_Transcript pending (part ${n})._`,
104
+ unavailable: `_Transcription unavailable (part ${n})._`,
105
+ };
106
+ }
107
+
108
+ /**
109
+ * A valid segment index (integer ≥ 0) off attachment metadata, else
110
+ * `undefined` — the un-segmented path. Client-set at link time; anything that
111
+ * isn't a non-negative integer falls back to the bare markers rather than
112
+ * fabricating a `(part N)`.
113
+ */
114
+ function segmentIndexOf(meta: { segment_index?: unknown }): number | undefined {
115
+ const raw = meta.segment_index;
116
+ return typeof raw === "number" && Number.isInteger(raw) && raw >= 0 ? raw : undefined;
117
+ }
102
118
 
103
119
  /**
104
120
  * Default sweep cadence (ms). The sweep is the safety net for backoff-
@@ -195,6 +211,14 @@ interface PendingMeta {
195
211
  * worker preserves the original stub-patching behavior (Lens flow).
196
212
  */
197
213
  transcribe_origin?: "auto" | "legacy";
214
+ /**
215
+ * Voice W2 (segmented recordings): a client-set 0-based index marking this
216
+ * attachment as one segment of a longer recording sliced into ~10-min parts,
217
+ * all linked on ONE note. When present, the legacy in-body path targets this
218
+ * part's markers (`… (part N)._`, N = segment_index + 1) rather than the bare
219
+ * ones — making per-part ordering structurally guaranteed. See `markersFor`.
220
+ */
221
+ segment_index?: number;
198
222
  [k: string]: unknown;
199
223
  }
200
224
 
@@ -357,17 +381,28 @@ export function startTranscriptionWorker(opts: TranscriptionWorkerOpts): Transcr
357
381
  * attachment failure we're trying to record.
358
382
  *
359
383
  * Body policy (finding F — never destroy content):
360
- * - Placeholder PRESENT → surgical replace of `_Transcript pending._`
361
- * with the marker. The `![[memo]]` embed + any surrounding text survive.
362
- * - Marker ALREADY PRESENT no-op (idempotent; a double-terminal-failure
363
- * must not stack markers).
364
- * - Otherwise (placeholder absent the user edited the note while it was
365
- * pending) APPEND `\n\n` + marker to the existing content. The old
366
- * code full-replaced the body here, destroying the embed AND the user's
367
- * edits. We append instead so nothing is lost. If the content is empty,
368
- * the marker alone becomes the body (avoids a leading blank line).
384
+ * - Pending marker PRESENT → surgical replace of the pending marker with
385
+ * the failure marker. The `![[memo]]` embed + any surrounding text
386
+ * survive. For a segmented attachment (`segment_index` set) this is the
387
+ * per-part `_Transcript pending (part N)._`; otherwise the bare marker.
388
+ * - Failure marker ALREADY PRESENT no-op (idempotent; a double-terminal-
389
+ * failure must not stack markers).
390
+ * - Otherwise (pending marker absent the user edited the note while it
391
+ * was pending) APPEND `\n\n` + failure marker to the existing content.
392
+ * The old code full-replaced the body here, destroying the embed AND the
393
+ * user's edits. We append instead so nothing is lost. If the content is
394
+ * empty, the marker alone becomes the body (avoids a leading blank line).
369
395
  */
370
- async function applyFailureMarker(store: Store, noteId: string): Promise<void> {
396
+ async function applyFailureMarker(
397
+ store: Store,
398
+ noteId: string,
399
+ segmentIndex: number | undefined,
400
+ ): Promise<void> {
401
+ // Bare markers by default; this segment's `(part N)` markers when the
402
+ // attachment carries a `segment_index` (voice W2). String-search replace
403
+ // targets the FIRST occurrence (a canonical body holds exactly one), and
404
+ // the includes-guard below keeps a repeated terminal failure from stacking.
405
+ const { pending, unavailable } = markersFor(segmentIndex);
371
406
  // OC-guarded (vault#435): the read-transform-write below is re-run against
372
407
  // fresh content on a conflict so a concurrent user edit isn't clobbered.
373
408
  // The transform is pure w.r.t. the note it's handed; the stub-set and
@@ -381,17 +416,24 @@ export function startTranscriptionWorker(opts: TranscriptionWorkerOpts): Transcr
381
416
  if (noteMeta.transcribe_stub !== true) return null;
382
417
 
383
418
  let body: string;
384
- if (TRANSCRIPT_PLACEHOLDER.test(note.content)) {
385
- body = note.content.replace(TRANSCRIPT_PLACEHOLDER, TRANSCRIPT_UNAVAILABLE);
386
- } else if (note.content.includes(TRANSCRIPT_UNAVAILABLE)) {
419
+ if (note.content.includes(pending)) {
420
+ // Function replacer so the search string is treated literally and the
421
+ // (fixed) failure marker is inserted verbatim.
422
+ body = note.content.replace(pending, () => unavailable);
423
+ } else if (note.content.includes(unavailable)) {
387
424
  // Marker already present — nothing to do. Clear the stub and
388
425
  // return without rewriting the body so we don't stack markers.
389
426
  body = note.content;
390
427
  } else {
391
428
  body = note.content.length > 0
392
- ? `${note.content}\n\n${TRANSCRIPT_UNAVAILABLE}`
393
- : TRANSCRIPT_UNAVAILABLE;
429
+ ? `${note.content}\n\n${unavailable}`
430
+ : unavailable;
394
431
  }
432
+ // Segmented: the stub is SHARED across this note's parts — keep it set
433
+ // so sibling parts still resolve their own slots. Return content only
434
+ // (leave note metadata untouched). Un-segmented: clear the one-shot
435
+ // stub as before (byte-unchanged).
436
+ if (segmentIndex !== undefined) return { content: body };
395
437
  const { transcribe_stub: _drop, ...restMeta } = noteMeta;
396
438
  return { content: body, metadata: restMeta };
397
439
  },
@@ -449,6 +491,12 @@ export function startTranscriptionWorker(opts: TranscriptionWorkerOpts): Transcr
449
491
  // vs. the legacy stub-patching path (Lens flow). Auto-write notes also
450
492
  // surface failures so the user can retry from the transcript note.
451
493
  const isAutoOrigin = meta.transcribe_origin === "auto";
494
+ // Voice W2: when this attachment is one segment of a longer recording
495
+ // (client-set `segment_index`), the legacy in-body path targets this
496
+ // part's markers instead of the bare ones. Undefined for un-segmented
497
+ // attachments — byte-unchanged behavior. Only the legacy path consults it;
498
+ // the auto/transcript-note path is untouched (segments are a memo concern).
499
+ const segmentIndex = segmentIndexOf(meta);
452
500
 
453
501
  // Honor backoff — we re-check here in case another tick queued this
454
502
  // attachment between the listing and now.
@@ -469,7 +517,7 @@ export function startTranscriptionWorker(opts: TranscriptionWorkerOpts): Transcr
469
517
  if (isAutoOrigin) {
470
518
  await writeFailureTranscriptNote(store, attachment, "audio file not found", undefined, undefined);
471
519
  } else {
472
- await applyFailureMarker(store, attachment.noteId);
520
+ await applyFailureMarker(store, attachment.noteId, segmentIndex);
473
521
  }
474
522
  return;
475
523
  }
@@ -527,7 +575,7 @@ export function startTranscriptionWorker(opts: TranscriptionWorkerOpts): Transcr
527
575
  if (isAutoOrigin) {
528
576
  await writeFailureTranscriptNote(store, attachment, errMsg, apiErr?.code, undefined);
529
577
  } else {
530
- await applyFailureMarker(store, attachment.noteId);
578
+ await applyFailureMarker(store, attachment.noteId, segmentIndex);
531
579
  }
532
580
  // retention=never drops the audio on any terminal state, including
533
581
  // failure. The user opted in to "I don't want the audio kept around
@@ -578,6 +626,16 @@ export function startTranscriptionWorker(opts: TranscriptionWorkerOpts): Transcr
578
626
  // before the transcript arrives opts out of the overwrite. OC-guarded
579
627
  // (vault#435): re-applied against fresh content on a conflict so a
580
628
  // concurrent user edit isn't clobbered.
629
+ //
630
+ // Success replaces whichever of THIS part's markers is present (bare, or
631
+ // `(part N)` when segmented). Built with no `/g` flag so `.replace`
632
+ // swaps only the FIRST match — a canonical capture body holds exactly
633
+ // one marker per part; alternation preserves positional-first semantics
634
+ // (a retried success replaces the failure marker where a first-try
635
+ // success replaced the pending one) so byte-for-byte matching today's
636
+ // un-segmented behavior.
637
+ const { pending, unavailable } = markersFor(segmentIndex);
638
+ const successTarget = new RegExp(`${escapeRegExp(pending)}|${escapeRegExp(unavailable)}`);
581
639
  await applyNoteTransformWithOC(
582
640
  store,
583
641
  attachment.noteId,
@@ -586,28 +644,31 @@ export function startTranscriptionWorker(opts: TranscriptionWorkerOpts): Transcr
586
644
  const noteMeta = (note.metadata as Record<string, unknown> | undefined) ?? {};
587
645
  if (noteMeta.transcribe_stub !== true) return null;
588
646
  // Body policy (finding F — never destroy content):
589
- // - placeholder OR failure-marker present → surgical replace in
590
- // place (a retried success replaces the `_Transcription
591
- // unavailable._` marker, landing exactly where a first-try
592
- // success would). The embed + surrounding capture body survive.
647
+ // - pending OR failure marker present → surgical replace in place.
648
+ // The embed + surrounding capture body survive.
593
649
  // - neither present (user edited the note while pending) → APPEND
594
650
  // the transcript instead of full-replacing the body, so the
595
651
  // user's edits + the `![[memo]]` embed are preserved. The old
596
652
  // code full-replaced here, which destroyed both.
597
653
  let body: string;
598
- if (TRANSCRIPT_SUCCESS_TARGET.test(note.content)) {
654
+ if (successTarget.test(note.content)) {
599
655
  // Function replacer, NOT a string — speech-to-text is arbitrary
600
656
  // user content, and String.replace treats `$&`, `$\``, `$'`,
601
657
  // `$1`-`$9` as special patterns in a string replacement. A
602
658
  // transcript containing `$&` would otherwise inject the matched
603
659
  // marker text into the body. `() => transcript` returns the text
604
660
  // verbatim.
605
- body = note.content.replace(TRANSCRIPT_SUCCESS_TARGET, () => transcript);
661
+ body = note.content.replace(successTarget, () => transcript);
606
662
  } else {
607
663
  body = note.content.length > 0
608
664
  ? `${note.content}\n\n${transcript}`
609
665
  : transcript;
610
666
  }
667
+ // Segmented: the stub is SHARED across this note's parts — keep it
668
+ // set so sibling parts still resolve their own slots. Return content
669
+ // only (leave note metadata untouched). Un-segmented: clear the
670
+ // one-shot stub as before (byte-unchanged).
671
+ if (segmentIndex !== undefined) return { content: body };
611
672
  const { transcribe_stub: _drop, ...restMeta } = noteMeta;
612
673
  return { content: body, metadata: restMeta };
613
674
  },
package/src/vault.test.ts CHANGED
@@ -6625,12 +6625,15 @@ describe("stateless MCP transport", async () => {
6625
6625
  expect(toolNames).not.toContain("merge-tags");
6626
6626
  // Admin tools (vault#376) are hidden too
6627
6627
  expect(toolNames).not.toContain("manage-token");
6628
- // request-attachment-download is read-tier (upload is write-tier).
6628
+ // request-attachment-download and read-attachment are read-tier
6629
+ // (upload is write-tier).
6629
6630
  expect(toolNames).toContain("request-attachment-download");
6630
6631
  expect(toolNames).not.toContain("request-attachment-upload");
6631
- // Read tier is exactly 6 tools (doctor added by the re-tier;
6632
- // request-attachment-download added by the attachment-tickets design).
6633
- expect(toolNames.length).toBe(6);
6632
+ expect(toolNames).toContain("read-attachment");
6633
+ // Read tier is exactly 7 tools (doctor added by the re-tier;
6634
+ // request-attachment-download by the attachment-tickets design;
6635
+ // read-attachment by Wave 2).
6636
+ expect(toolNames.length).toBe(7);
6634
6637
 
6635
6638
  closeAllStores();
6636
6639
  });
@@ -6911,15 +6914,23 @@ describe("MCP tools/list scope tiers (vault#376)", () => {
6911
6914
  return names;
6912
6915
  }
6913
6916
 
6914
- test("vault:read sees exactly the 6 read tools (doctor moved admin → read; request-attachment-download is read-tier)", async () => {
6917
+ test("vault:read sees exactly the 7 read tools (doctor moved admin → read; read-attachment and request-attachment-download are read-tier)", async () => {
6915
6918
  const names = await listToolNames(["vault:read"]);
6916
6919
  expect(new Set(names)).toEqual(
6917
- new Set(["query-notes", "list-tags", "find-path", "vault-info", "doctor", "request-attachment-download"]),
6920
+ new Set([
6921
+ "query-notes",
6922
+ "list-tags",
6923
+ "find-path",
6924
+ "vault-info",
6925
+ "doctor",
6926
+ "request-attachment-download",
6927
+ "read-attachment",
6928
+ ]),
6918
6929
  );
6919
- expect(names.length).toBe(6);
6930
+ expect(names.length).toBe(7);
6920
6931
  });
6921
6932
 
6922
- test("vault:read + vault:write sees the 10 read+write tools (tag-schema tools moved write → admin; request-attachment-upload is write-tier)", async () => {
6933
+ test("vault:read + vault:write sees the 11 read+write tools (tag-schema tools moved write → admin; request-attachment-upload is write-tier)", async () => {
6923
6934
  const names = await listToolNames(["vault:read", "vault:write"]);
6924
6935
  expect(new Set(names)).toEqual(
6925
6936
  new Set([
@@ -6933,9 +6944,10 @@ describe("MCP tools/list scope tiers (vault#376)", () => {
6933
6944
  "delete-note",
6934
6945
  "request-attachment-upload",
6935
6946
  "request-attachment-download",
6947
+ "read-attachment",
6936
6948
  ]),
6937
6949
  );
6938
- expect(names.length).toBe(10);
6950
+ expect(names.length).toBe(11);
6939
6951
  expect(names).not.toContain("manage-token");
6940
6952
  // Re-tier (this PR): update-tag/delete-tag/rename-tag/merge-tags are now
6941
6953
  // admin-tier — structure/taxonomy curation, not content authorship.
@@ -6947,7 +6959,7 @@ describe("MCP tools/list scope tiers (vault#376)", () => {
6947
6959
  expect(names).toContain("delete-note");
6948
6960
  });
6949
6961
 
6950
- test("vault:admin sees all 16 tools including manage-token + prune-schema + the tag-schema tools + both attachment-ticket tools", async () => {
6962
+ test("vault:admin sees all 17 tools including manage-token + prune-schema + the tag-schema tools + all three attachment tools", async () => {
6951
6963
  const names = await listToolNames(["vault:read", "vault:write", "vault:admin"]);
6952
6964
  expect(names).toContain("manage-token");
6953
6965
  expect(names).toContain("prune-schema");
@@ -6958,10 +6970,11 @@ describe("MCP tools/list scope tiers (vault#376)", () => {
6958
6970
  expect(names).toContain("merge-tags");
6959
6971
  expect(names).toContain("request-attachment-upload");
6960
6972
  expect(names).toContain("request-attachment-download");
6961
- expect(names.length).toBe(16);
6973
+ expect(names).toContain("read-attachment");
6974
+ expect(names.length).toBe(17);
6962
6975
  });
6963
6976
 
6964
- test("legacy-derived full token sees all 16 tools (back-compat)", async () => {
6977
+ test("legacy-derived full token sees all 17 tools (back-compat)", async () => {
6965
6978
  const { handleScopedMcp } = await import("./mcp-http.ts");
6966
6979
  const { writeVaultConfig } = await import("./config.ts");
6967
6980
  const { closeAllStores } = await import("./vault-store.ts");
@@ -6994,7 +7007,7 @@ describe("MCP tools/list scope tiers (vault#376)", () => {
6994
7007
  } as any);
6995
7008
  const body = await res.json() as any;
6996
7009
  const names: string[] = body.result.tools.map((t: any) => t.name);
6997
- expect(names.length).toBe(16);
7010
+ expect(names.length).toBe(17);
6998
7011
  expect(names).toContain("manage-token");
6999
7012
  expect(names).toContain("prune-schema");
7000
7013
  expect(names).toContain("doctor");