@openparachute/vault 0.7.3-rc.12 → 0.7.3-rc.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +3 -5
- package/core/src/conformance.ts +1 -2
- package/core/src/content-range.test.ts +0 -127
- package/core/src/content-range.ts +0 -100
- package/core/src/contract-typed-index.test.ts +3 -4
- package/core/src/core.test.ts +4 -521
- package/core/src/expand.ts +3 -11
- package/core/src/indexed-fields.test.ts +3 -9
- package/core/src/indexed-fields.ts +1 -9
- package/core/src/mcp.ts +10 -601
- package/core/src/notes.ts +4 -201
- package/core/src/schema-defaults.ts +1 -85
- package/core/src/search-fts-v25.test.ts +1 -9
- package/core/src/search-query.test.ts +0 -42
- package/core/src/search-query.ts +0 -27
- package/core/src/seed-packs.test.ts +1 -117
- package/core/src/seed-packs.ts +1 -217
- package/core/src/store.ts +3 -59
- package/core/src/tag-schemas.ts +12 -27
- package/core/src/types.ts +1 -7
- package/core/src/vault-projection.ts +0 -55
- package/package.json +1 -1
- package/src/add-pack.test.ts +0 -32
- package/src/cli.ts +1 -5
- package/src/contract-errors.test.ts +1 -2
- package/src/mcp-http.ts +4 -33
- package/src/mcp-tools.ts +14 -61
- package/src/onboarding-seed.test.ts +0 -64
- package/src/routes.ts +95 -92
- package/src/routing.ts +0 -17
- package/src/server.ts +0 -7
- package/src/storage.test.ts +1 -200
- package/src/transcription-worker.test.ts +0 -151
- package/src/transcription-worker.ts +52 -113
- package/src/vault.test.ts +11 -48
- package/core/src/attachment/bytes-provider.ts +0 -65
- package/core/src/attachment/policy.test.ts +0 -66
- package/core/src/attachment/policy.ts +0 -131
- package/core/src/attachment/tickets.test.ts +0 -45
- package/core/src/attachment/tickets.ts +0 -117
- package/core/src/attachment-tickets-tool.test.ts +0 -286
- package/core/src/display-title.test.ts +0 -190
- package/core/src/lede.test.ts +0 -96
- package/core/src/search-title-boost.test.ts +0 -125
- package/src/attachment-bytes.ts +0 -68
- package/src/attachment-tickets.test.ts +0 -475
- package/src/attachment-tickets.ts +0 -340
- package/src/read-attachment.test.ts +0 -436
package/src/storage.test.ts
CHANGED
|
@@ -30,7 +30,7 @@ const testDir = join(
|
|
|
30
30
|
process.env.PARACHUTE_HOME = testDir;
|
|
31
31
|
process.env.ASSETS_DIR = join(testDir, "assets");
|
|
32
32
|
|
|
33
|
-
const { handleStorage, MAX_UPLOAD_BYTES, MAX_REQUEST_BODY_BYTES
|
|
33
|
+
const { handleStorage, MAX_UPLOAD_BYTES, MAX_REQUEST_BODY_BYTES } = await import("./routes.ts");
|
|
34
34
|
const { expandTokenTagScope } = await import("./tag-scope.ts");
|
|
35
35
|
|
|
36
36
|
// The upload-allowlist tests never touch the store (POST /upload writes to
|
|
@@ -564,202 +564,3 @@ describe("storage POST parity (finding D)", () => {
|
|
|
564
564
|
expect(res.status).toBe(404);
|
|
565
565
|
});
|
|
566
566
|
});
|
|
567
|
-
|
|
568
|
-
// ---------------------------------------------------------------------------
|
|
569
|
-
// REST Range (206) — vault attachments-for-agents design D9, the REST twin
|
|
570
|
-
// of MCP's content_offset/content_length.
|
|
571
|
-
// ---------------------------------------------------------------------------
|
|
572
|
-
|
|
573
|
-
describe("parseByteRangeHeader", () => {
|
|
574
|
-
test("no header → null (full response)", () => {
|
|
575
|
-
expect(parseByteRangeHeader(null, 100)).toBeNull();
|
|
576
|
-
});
|
|
577
|
-
|
|
578
|
-
test("bytes=0-99 on a 100-byte file → the whole file as one satisfiable range", () => {
|
|
579
|
-
expect(parseByteRangeHeader("bytes=0-99", 100)).toEqual({ start: 0, end: 99 });
|
|
580
|
-
});
|
|
581
|
-
|
|
582
|
-
test("bytes=10-19 → an inclusive mid-file window", () => {
|
|
583
|
-
expect(parseByteRangeHeader("bytes=10-19", 100)).toEqual({ start: 10, end: 19 });
|
|
584
|
-
});
|
|
585
|
-
|
|
586
|
-
test("bytes=90- (open-ended) → reads to EOF", () => {
|
|
587
|
-
expect(parseByteRangeHeader("bytes=90-", 100)).toEqual({ start: 90, end: 99 });
|
|
588
|
-
});
|
|
589
|
-
|
|
590
|
-
test("bytes=-10 (suffix range) → the last 10 bytes", () => {
|
|
591
|
-
expect(parseByteRangeHeader("bytes=-10", 100)).toEqual({ start: 90, end: 99 });
|
|
592
|
-
});
|
|
593
|
-
|
|
594
|
-
test("bytes=-1000 (suffix longer than the file) → clamps to the whole file", () => {
|
|
595
|
-
expect(parseByteRangeHeader("bytes=-1000", 100)).toEqual({ start: 0, end: 99 });
|
|
596
|
-
});
|
|
597
|
-
|
|
598
|
-
test("bytes=50-1000 (end past EOF) → clamps end to the last byte", () => {
|
|
599
|
-
expect(parseByteRangeHeader("bytes=50-1000", 100)).toEqual({ start: 50, end: 99 });
|
|
600
|
-
});
|
|
601
|
-
|
|
602
|
-
test("start past EOF → null (this function's own contract; a real Bun.serve() socket overrides with a native 416 — see doc comment)", () => {
|
|
603
|
-
expect(parseByteRangeHeader("bytes=100-200", 100)).toBeNull();
|
|
604
|
-
expect(parseByteRangeHeader("bytes=1000-", 100)).toBeNull();
|
|
605
|
-
});
|
|
606
|
-
|
|
607
|
-
test("start > end → null (malformed)", () => {
|
|
608
|
-
expect(parseByteRangeHeader("bytes=50-10", 100)).toBeNull();
|
|
609
|
-
});
|
|
610
|
-
|
|
611
|
-
test("bytes=- (both empty) → null", () => {
|
|
612
|
-
expect(parseByteRangeHeader("bytes=-", 100)).toBeNull();
|
|
613
|
-
});
|
|
614
|
-
|
|
615
|
-
test("a multi-range list (bytes=0-10,20-30) → null (ignored per D9, not an error)", () => {
|
|
616
|
-
expect(parseByteRangeHeader("bytes=0-10,20-30", 100)).toBeNull();
|
|
617
|
-
});
|
|
618
|
-
|
|
619
|
-
test("an unrecognized unit → null", () => {
|
|
620
|
-
expect(parseByteRangeHeader("items=0-10", 100)).toBeNull();
|
|
621
|
-
});
|
|
622
|
-
|
|
623
|
-
test("garbage value → null, not a throw", () => {
|
|
624
|
-
expect(parseByteRangeHeader("bytes=abc-def", 100)).toBeNull();
|
|
625
|
-
expect(parseByteRangeHeader("nonsense", 100)).toBeNull();
|
|
626
|
-
});
|
|
627
|
-
|
|
628
|
-
test("a zero-byte file → null regardless of header (nothing to range over)", () => {
|
|
629
|
-
expect(parseByteRangeHeader("bytes=0-0", 0)).toBeNull();
|
|
630
|
-
});
|
|
631
|
-
});
|
|
632
|
-
|
|
633
|
-
describe("storage GET Range support (vault attachments-for-agents design D9)", () => {
|
|
634
|
-
const VAULT = "range-vault";
|
|
635
|
-
// 26 bytes, byte-identical to its own index — content[i] === i (0x00..0x19)
|
|
636
|
-
// — so any slice's bytes can be asserted exactly by index, not just length.
|
|
637
|
-
const CONTENT = Buffer.from(Array.from({ length: 26 }, (_, i) => i));
|
|
638
|
-
|
|
639
|
-
async function setup(): Promise<{ store: SqliteStore; assets: string; relPath: string }> {
|
|
640
|
-
const store = freshStore();
|
|
641
|
-
const assets = join(testDir, "assets", VAULT, "data");
|
|
642
|
-
mkdirSync(join(assets, "2026-05-28"), { recursive: true });
|
|
643
|
-
process.env.ASSETS_DIR = assets;
|
|
644
|
-
|
|
645
|
-
const relPath = "2026-05-28/ranged.bin";
|
|
646
|
-
writeFileSync(join(assets, relPath), CONTENT);
|
|
647
|
-
const note = await store.createNote("range note", { tags: ["misc"] });
|
|
648
|
-
await store.addAttachment(note.id, relPath, "application/octet-stream");
|
|
649
|
-
|
|
650
|
-
return { store, assets, relPath };
|
|
651
|
-
}
|
|
652
|
-
|
|
653
|
-
function getReq(reqPath: string, range?: string): Request {
|
|
654
|
-
const headers: Record<string, string> = {};
|
|
655
|
-
if (range !== undefined) headers.range = range;
|
|
656
|
-
return new Request(`http://localhost:1940/storage/${reqPath}`, { method: "GET", headers });
|
|
657
|
-
}
|
|
658
|
-
|
|
659
|
-
test("no Range header → 200, full body, Accept-Ranges advertised", async () => {
|
|
660
|
-
const { store, relPath } = await setup();
|
|
661
|
-
const res = await handleStorage(getReq(relPath), `/${relPath}`, VAULT, store);
|
|
662
|
-
expect(res.status).toBe(200);
|
|
663
|
-
expect(res.headers.get("Accept-Ranges")).toBe("bytes");
|
|
664
|
-
expect(res.headers.get("Content-Length")).toBe("26");
|
|
665
|
-
expect(res.headers.has("Content-Range")).toBe(false);
|
|
666
|
-
const body = Buffer.from(await res.arrayBuffer());
|
|
667
|
-
expect(body.equals(CONTENT)).toBe(true);
|
|
668
|
-
});
|
|
669
|
-
|
|
670
|
-
test("Range: bytes=0-9 → 206, first 10 bytes, correct Content-Range", async () => {
|
|
671
|
-
const { store, relPath } = await setup();
|
|
672
|
-
const res = await handleStorage(getReq(relPath, "bytes=0-9"), `/${relPath}`, VAULT, store);
|
|
673
|
-
expect(res.status).toBe(206);
|
|
674
|
-
expect(res.headers.get("Content-Range")).toBe("bytes 0-9/26");
|
|
675
|
-
expect(res.headers.get("Content-Length")).toBe("10");
|
|
676
|
-
expect(res.headers.get("Accept-Ranges")).toBe("bytes");
|
|
677
|
-
expect(res.headers.get("X-Content-Type-Options")).toBe("nosniff");
|
|
678
|
-
const body = Buffer.from(await res.arrayBuffer());
|
|
679
|
-
expect(body.equals(CONTENT.subarray(0, 10))).toBe(true);
|
|
680
|
-
});
|
|
681
|
-
|
|
682
|
-
test("Range: bytes=16-25 (to EOF) → 206, correct tail bytes", async () => {
|
|
683
|
-
const { store, relPath } = await setup();
|
|
684
|
-
const res = await handleStorage(getReq(relPath, "bytes=16-25"), `/${relPath}`, VAULT, store);
|
|
685
|
-
expect(res.status).toBe(206);
|
|
686
|
-
expect(res.headers.get("Content-Range")).toBe("bytes 16-25/26");
|
|
687
|
-
const body = Buffer.from(await res.arrayBuffer());
|
|
688
|
-
expect(body.equals(CONTENT.subarray(16, 26))).toBe(true);
|
|
689
|
-
});
|
|
690
|
-
|
|
691
|
-
test("Range: bytes=-5 (suffix) → 206, last 5 bytes", async () => {
|
|
692
|
-
const { store, relPath } = await setup();
|
|
693
|
-
const res = await handleStorage(getReq(relPath, "bytes=-5"), `/${relPath}`, VAULT, store);
|
|
694
|
-
expect(res.status).toBe(206);
|
|
695
|
-
expect(res.headers.get("Content-Range")).toBe("bytes 21-25/26");
|
|
696
|
-
const body = Buffer.from(await res.arrayBuffer());
|
|
697
|
-
expect(body.equals(CONTENT.subarray(21, 26))).toBe(true);
|
|
698
|
-
});
|
|
699
|
-
|
|
700
|
-
test("paging the whole file via sequential ranges reassembles byte-identical content", async () => {
|
|
701
|
-
const { store, relPath } = await setup();
|
|
702
|
-
const chunkSize = 7; // doesn't divide 26 evenly — exercises a short final chunk
|
|
703
|
-
const chunks: Buffer[] = [];
|
|
704
|
-
for (let start = 0; start < 26; start += chunkSize) {
|
|
705
|
-
const end = Math.min(start + chunkSize - 1, 25);
|
|
706
|
-
const res = await handleStorage(getReq(relPath, `bytes=${start}-${end}`), `/${relPath}`, VAULT, store);
|
|
707
|
-
expect(res.status).toBe(206);
|
|
708
|
-
chunks.push(Buffer.from(await res.arrayBuffer()));
|
|
709
|
-
}
|
|
710
|
-
expect(Buffer.concat(chunks).equals(CONTENT)).toBe(true);
|
|
711
|
-
});
|
|
712
|
-
|
|
713
|
-
test("a malformed Range header → 200 full response, not an error (D9: ignore, don't fail)", async () => {
|
|
714
|
-
const { store, relPath } = await setup();
|
|
715
|
-
const res = await handleStorage(getReq(relPath, "not-a-range"), `/${relPath}`, VAULT, store);
|
|
716
|
-
expect(res.status).toBe(200);
|
|
717
|
-
expect(res.headers.has("Content-Range")).toBe(false);
|
|
718
|
-
const body = Buffer.from(await res.arrayBuffer());
|
|
719
|
-
expect(body.equals(CONTENT)).toBe(true);
|
|
720
|
-
});
|
|
721
|
-
|
|
722
|
-
test("a multi-range Range header → 200 full response (multipart ranges unsupported, ignored per D9)", async () => {
|
|
723
|
-
const { store, relPath } = await setup();
|
|
724
|
-
const res = await handleStorage(getReq(relPath, "bytes=0-5,10-15"), `/${relPath}`, VAULT, store);
|
|
725
|
-
expect(res.status).toBe(200);
|
|
726
|
-
const body = Buffer.from(await res.arrayBuffer());
|
|
727
|
-
expect(body.equals(CONTENT)).toBe(true);
|
|
728
|
-
});
|
|
729
|
-
|
|
730
|
-
test("an unsatisfiable Range (start past EOF) → handleStorage's OWN logic returns 200 (parseByteRangeHeader's null contract)", async () => {
|
|
731
|
-
// IMPORTANT — this is NOT the whole live story. `handleStorage` is
|
|
732
|
-
// called here directly, in-process, so this only exercises OUR
|
|
733
|
-
// fallback logic (parseByteRangeHeader → null → the unranged 200
|
|
734
|
-
// branch). Live-verified against a real `Bun.serve()` socket (not
|
|
735
|
-
// this harness): Bun's own runtime independently reinterprets the
|
|
736
|
-
// incoming `Range` header on a `Bun.file()`-backed response body and
|
|
737
|
-
// overrides an out-of-bounds range with a native
|
|
738
|
-
// **416 Range Not Satisfiable** — RFC 7233-correct, and kept, not
|
|
739
|
-
// fought. This in-process harness has no socket for Bun's native
|
|
740
|
-
// layer to intercept, so it can only ever observe the 200 our own
|
|
741
|
-
// code returns — see `parseByteRangeHeader`'s doc comment
|
|
742
|
-
// (`src/routes.ts`) for the full explanation.
|
|
743
|
-
const { store, relPath } = await setup();
|
|
744
|
-
const res = await handleStorage(getReq(relPath, "bytes=1000-2000"), `/${relPath}`, VAULT, store);
|
|
745
|
-
expect(res.status).toBe(200);
|
|
746
|
-
const body = Buffer.from(await res.arrayBuffer());
|
|
747
|
-
expect(body.equals(CONTENT)).toBe(true);
|
|
748
|
-
});
|
|
749
|
-
|
|
750
|
-
test("Range is still honored for a tag-scoped in-scope request (206)", async () => {
|
|
751
|
-
const { store, relPath } = await setup();
|
|
752
|
-
const ctx = await tagScopeCtx(store, ["misc"]);
|
|
753
|
-
const res = await handleStorage(getReq(relPath, "bytes=0-3"), `/${relPath}`, VAULT, store, ctx);
|
|
754
|
-
expect(res.status).toBe(206);
|
|
755
|
-
const body = Buffer.from(await res.arrayBuffer());
|
|
756
|
-
expect(body.equals(CONTENT.subarray(0, 4))).toBe(true);
|
|
757
|
-
});
|
|
758
|
-
|
|
759
|
-
test("Range on an out-of-scope attachment still 404s (same confinement guard, byte-identical to the non-ranged path)", async () => {
|
|
760
|
-
const { store, relPath } = await setup();
|
|
761
|
-
const ctx = await tagScopeCtx(store, ["totally-unrelated-tag"]);
|
|
762
|
-
const res = await handleStorage(getReq(relPath, "bytes=0-3"), `/${relPath}`, VAULT, store, ctx);
|
|
763
|
-
expect(res.status).toBe(404);
|
|
764
|
-
});
|
|
765
|
-
});
|
|
@@ -1627,154 +1627,3 @@ describe("transcription worker — legacy in-body memo content safety (finding F
|
|
|
1627
1627
|
expect(note!.content).toContain("retry text with $& and $0 and $$ literal");
|
|
1628
1628
|
});
|
|
1629
1629
|
});
|
|
1630
|
-
|
|
1631
|
-
// ---------------------------------------------------------------------------
|
|
1632
|
-
// voice W2 — segmented recordings. The app slices a long recording into
|
|
1633
|
-
// ~10-min segments, each linked as its own attachment on ONE note carrying a
|
|
1634
|
-
// client-set `segment_index` (0-based). The legacy in-body path targets that
|
|
1635
|
-
// part's markers — `_Transcript pending (part N)._` / `_Transcription
|
|
1636
|
-
// unavailable (part N)._`, N = segment_index + 1 — so each transcript lands in
|
|
1637
|
-
// its own pre-allocated slot regardless of completion order. Marker strings
|
|
1638
|
-
// are a BYTE-EXACT cross-door + cross-repo contract (the cloud worker ships
|
|
1639
|
-
// the identical text), so these assertions pin the exact bytes.
|
|
1640
|
-
// ---------------------------------------------------------------------------
|
|
1641
|
-
|
|
1642
|
-
describe("transcription worker — segmented recordings (voice W2)", () => {
|
|
1643
|
-
test("three parts completing OUT OF ORDER (2 ok, 0 fails, 1 ok) each land in their own slot", async () => {
|
|
1644
|
-
// One note, three pre-allocated part slots + the shared stub opt-in.
|
|
1645
|
-
const body =
|
|
1646
|
-
"# 🎙️ Voice memo\n\n" +
|
|
1647
|
-
"_Transcript pending (part 1)._\n\n" +
|
|
1648
|
-
"_Transcript pending (part 2)._\n\n" +
|
|
1649
|
-
"_Transcript pending (part 3)._\n";
|
|
1650
|
-
await store.createNote(body, { id: "seg-note", metadata: { transcribe_stub: true } });
|
|
1651
|
-
|
|
1652
|
-
seedAudio("memos/seg0.webm");
|
|
1653
|
-
seedAudio("memos/seg1.webm");
|
|
1654
|
-
seedAudio("memos/seg2.webm");
|
|
1655
|
-
const seg0 = await store.addAttachment("seg-note", "memos/seg0.webm", "audio/webm", {
|
|
1656
|
-
transcribe_status: "pending",
|
|
1657
|
-
segment_index: 0,
|
|
1658
|
-
});
|
|
1659
|
-
const seg1 = await store.addAttachment("seg-note", "memos/seg1.webm", "audio/webm", {
|
|
1660
|
-
transcribe_status: "pending",
|
|
1661
|
-
segment_index: 1,
|
|
1662
|
-
});
|
|
1663
|
-
const seg2 = await store.addAttachment("seg-note", "memos/seg2.webm", "audio/webm", {
|
|
1664
|
-
transcribe_status: "pending",
|
|
1665
|
-
segment_index: 2,
|
|
1666
|
-
});
|
|
1667
|
-
|
|
1668
|
-
// Complete part 3 (segment_index 2) FIRST — success.
|
|
1669
|
-
const w2 = makeWorker({ fetchImpl: mkFetchMock([{ text: "part three text" }]) });
|
|
1670
|
-
try { await w2.kick("default", seg2); } finally { await w2.stop(); }
|
|
1671
|
-
|
|
1672
|
-
// Then part 1 (segment_index 0) — terminal failure (maxAttempts=1).
|
|
1673
|
-
const w0 = makeWorker({
|
|
1674
|
-
fetchImpl: mkFetchMock([{ error: "scribe down", status: 500 }]),
|
|
1675
|
-
maxAttempts: 1,
|
|
1676
|
-
});
|
|
1677
|
-
try { await w0.kick("default", seg0); } finally { await w0.stop(); }
|
|
1678
|
-
|
|
1679
|
-
// Finally part 2 (segment_index 1) — success.
|
|
1680
|
-
const w1 = makeWorker({ fetchImpl: mkFetchMock([{ text: "part two text" }]) });
|
|
1681
|
-
try { await w1.kick("default", seg1); } finally { await w1.stop(); }
|
|
1682
|
-
|
|
1683
|
-
const note = await store.getNote("seg-note");
|
|
1684
|
-
// Each part landed in its own slot: part 1 → failure marker, part 2/3 →
|
|
1685
|
-
// their transcripts, despite completing 3 → 1 → 2.
|
|
1686
|
-
expect(note!.content).toBe(
|
|
1687
|
-
"# 🎙️ Voice memo\n\n" +
|
|
1688
|
-
"_Transcription unavailable (part 1)._\n\n" +
|
|
1689
|
-
"part two text\n\n" +
|
|
1690
|
-
"part three text\n",
|
|
1691
|
-
);
|
|
1692
|
-
// Shared stub SURVIVES — sibling parts each needed the gate open; a
|
|
1693
|
-
// per-part clear (the un-segmented behavior) would have blocked parts 2 & 3.
|
|
1694
|
-
expect((note!.metadata as any)?.transcribe_stub).toBe(true);
|
|
1695
|
-
|
|
1696
|
-
// Attachment rows reflect their individual outcomes.
|
|
1697
|
-
const atts = await store.getAttachments("seg-note");
|
|
1698
|
-
const byIndex = Object.fromEntries(atts.map((a) => [a.metadata?.segment_index, a]));
|
|
1699
|
-
expect(byIndex[0]!.metadata?.transcribe_status).toBe("failed");
|
|
1700
|
-
expect(byIndex[1]!.metadata?.transcribe_status).toBe("done");
|
|
1701
|
-
expect(byIndex[1]!.metadata?.transcript).toBe("part two text");
|
|
1702
|
-
expect(byIndex[2]!.metadata?.transcribe_status).toBe("done");
|
|
1703
|
-
expect(byIndex[2]!.metadata?.transcript).toBe("part three text");
|
|
1704
|
-
});
|
|
1705
|
-
|
|
1706
|
-
test("un-segmented attachment → BARE markers, one-shot stub cleared (regression pin)", async () => {
|
|
1707
|
-
// No `segment_index` → byte-for-byte the pre-W2 behavior: replace the bare
|
|
1708
|
-
// placeholder, clear the stub. A stray `(part N)` marker in the body is
|
|
1709
|
-
// left untouched (the bare path never targets part markers).
|
|
1710
|
-
await store.createNote(
|
|
1711
|
-
"# 🎙️ Voice memo\n\n_Transcript pending._\n\n_Transcript pending (part 2)._\n",
|
|
1712
|
-
{ id: "unseg-note", metadata: { transcribe_stub: true } },
|
|
1713
|
-
);
|
|
1714
|
-
seedAudio("memos/unseg.webm");
|
|
1715
|
-
const att = await store.addAttachment("unseg-note", "memos/unseg.webm", "audio/webm", {
|
|
1716
|
-
transcribe_status: "pending",
|
|
1717
|
-
});
|
|
1718
|
-
|
|
1719
|
-
const worker = makeWorker({ fetchImpl: mkFetchMock([{ text: "bare transcript" }]) });
|
|
1720
|
-
try { await worker.kick("default", att); } finally { await worker.stop(); }
|
|
1721
|
-
|
|
1722
|
-
const note = await store.getNote("unseg-note");
|
|
1723
|
-
// Bare marker replaced; the part-2 marker is NOT touched by the bare path.
|
|
1724
|
-
expect(note!.content).toBe(
|
|
1725
|
-
"# 🎙️ Voice memo\n\nbare transcript\n\n_Transcript pending (part 2)._\n",
|
|
1726
|
-
);
|
|
1727
|
-
// One-shot stub cleared, exactly as today.
|
|
1728
|
-
expect((note!.metadata as any)?.transcribe_stub).toBeUndefined();
|
|
1729
|
-
});
|
|
1730
|
-
|
|
1731
|
-
test("malformed segment_index falls back to bare markers (contract: integer ≥ 0)", async () => {
|
|
1732
|
-
// A non-integer / negative `segment_index` must NOT fabricate a `(part N)`
|
|
1733
|
-
// marker — it degrades to the bare path (fully backward compatible).
|
|
1734
|
-
await store.createNote(
|
|
1735
|
-
"# 🎙️ Voice memo\n\n_Transcript pending._\n",
|
|
1736
|
-
{ id: "seg-bad", metadata: { transcribe_stub: true } },
|
|
1737
|
-
);
|
|
1738
|
-
seedAudio("memos/segbad.webm");
|
|
1739
|
-
const att = await store.addAttachment("seg-bad", "memos/segbad.webm", "audio/webm", {
|
|
1740
|
-
transcribe_status: "pending",
|
|
1741
|
-
segment_index: -1, // invalid → bare path
|
|
1742
|
-
});
|
|
1743
|
-
|
|
1744
|
-
const worker = makeWorker({ fetchImpl: mkFetchMock([{ text: "fallback transcript" }]) });
|
|
1745
|
-
try { await worker.kick("default", att); } finally { await worker.stop(); }
|
|
1746
|
-
|
|
1747
|
-
const note = await store.getNote("seg-bad");
|
|
1748
|
-
expect(note!.content).toBe("# 🎙️ Voice memo\n\nfallback transcript\n");
|
|
1749
|
-
expect((note!.metadata as any)?.transcribe_stub).toBeUndefined();
|
|
1750
|
-
});
|
|
1751
|
-
|
|
1752
|
-
test("segmented part whose marker was edited away → transcript appended (graceful fallback)", async () => {
|
|
1753
|
-
// The user rewrote part 2's slot by hand, removing its pending marker.
|
|
1754
|
-
// Mirror the un-segmented replace-or-append policy scoped to the part:
|
|
1755
|
-
// with neither of part 2's markers present, append the transcript
|
|
1756
|
-
// (no `(part N)` prefix) rather than destroying the user's edit.
|
|
1757
|
-
const editedBody =
|
|
1758
|
-
"# 🎙️ Voice memo\n\n" +
|
|
1759
|
-
"_Transcript pending (part 1)._\n\n" +
|
|
1760
|
-
"User rewrote part two by hand.\n";
|
|
1761
|
-
await store.createNote(editedBody, { id: "seg-edit", metadata: { transcribe_stub: true } });
|
|
1762
|
-
seedAudio("memos/segedit.webm");
|
|
1763
|
-
const seg1 = await store.addAttachment("seg-edit", "memos/segedit.webm", "audio/webm", {
|
|
1764
|
-
transcribe_status: "pending",
|
|
1765
|
-
segment_index: 1, // part 2
|
|
1766
|
-
});
|
|
1767
|
-
|
|
1768
|
-
const worker = makeWorker({ fetchImpl: mkFetchMock([{ text: "the real part two" }]) });
|
|
1769
|
-
try { await worker.kick("default", seg1); } finally { await worker.stop(); }
|
|
1770
|
-
|
|
1771
|
-
const note = await store.getNote("seg-edit");
|
|
1772
|
-
// Part 2's transcript appended; the user's edit + part 1's pending slot
|
|
1773
|
-
// both survive untouched.
|
|
1774
|
-
expect(note!.content).toBe(`${editedBody}\n\nthe real part two`);
|
|
1775
|
-
expect(note!.content).toContain("User rewrote part two by hand.");
|
|
1776
|
-
expect(note!.content).toContain("_Transcript pending (part 1)._");
|
|
1777
|
-
// Segmented → shared stub preserved (part 1 still needs the gate open).
|
|
1778
|
-
expect((note!.metadata as any)?.transcribe_stub).toBe(true);
|
|
1779
|
-
});
|
|
1780
|
-
});
|
|
@@ -64,57 +64,41 @@ import {
|
|
|
64
64
|
} from "../core/src/transcription/provider.ts";
|
|
65
65
|
import { ScribeHttpProvider } from "./transcription/providers/scribe-http.ts";
|
|
66
66
|
|
|
67
|
+
/** Placeholder pattern written by the voice-memo capture stub. */
|
|
68
|
+
const TRANSCRIPT_PLACEHOLDER = /_Transcript pending\._/;
|
|
69
|
+
|
|
67
70
|
/**
|
|
68
|
-
*
|
|
69
|
-
*
|
|
70
|
-
*
|
|
71
|
-
*
|
|
72
|
-
*
|
|
73
|
-
* transcription path ships the identical strings, and the notes-ui status
|
|
74
|
-
* chip (parachute-surface TranscriptionStatus.tsx) keys off the failure
|
|
75
|
-
* marker's exact copy. Don't change any of this text without a coordinated
|
|
76
|
-
* change in both places. A friendlier "retry available" copy + chip
|
|
77
|
-
* affordance is a tracked parachute-surface follow-up.
|
|
71
|
+
* Body written when transcription reaches a terminal failure (maxAttempts
|
|
72
|
+
* exhausted, or the audio file is missing). This used to be written by
|
|
73
|
+
* Lens's now-removed scribe client; owning it here means a failed upload
|
|
74
|
+
* stops reading "Transcript pending" forever regardless of which client
|
|
75
|
+
* uploaded the audio.
|
|
78
76
|
*
|
|
79
|
-
*
|
|
80
|
-
*
|
|
81
|
-
*
|
|
77
|
+
* NOTE: the notes-ui status chip (parachute-surface TranscriptionStatus.tsx)
|
|
78
|
+
* keys off this exact string, so don't change the copy without a coordinated
|
|
79
|
+
* change there. A friendlier "retry available" copy + chip affordance is a
|
|
80
|
+
* tracked parachute-surface follow-up.
|
|
82
81
|
*/
|
|
83
|
-
const
|
|
84
|
-
const BARE_UNAVAILABLE = "_Transcription unavailable._";
|
|
85
|
-
|
|
86
|
-
/** Escape a literal string for safe embedding in a `RegExp`. */
|
|
87
|
-
function escapeRegExp(literal: string): string {
|
|
88
|
-
return literal.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
89
|
-
}
|
|
82
|
+
const TRANSCRIPT_UNAVAILABLE = "_Transcription unavailable._";
|
|
90
83
|
|
|
91
84
|
/**
|
|
92
|
-
*
|
|
93
|
-
*
|
|
94
|
-
*
|
|
95
|
-
*
|
|
96
|
-
*
|
|
97
|
-
*
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
/**
|
|
109
|
-
* A valid segment index (integer ≥ 0) off attachment metadata, else
|
|
110
|
-
* `undefined` — the un-segmented path. Client-set at link time; anything that
|
|
111
|
-
* isn't a non-negative integer falls back to the bare markers rather than
|
|
112
|
-
* fabricating a `(part N)`.
|
|
85
|
+
* On a successful (re)transcription of a legacy in-body memo, the transcript
|
|
86
|
+
* replaces whichever marker is currently in the body — the original
|
|
87
|
+
* `_Transcript pending._` on a first-try success, OR `_Transcription
|
|
88
|
+
* unavailable._` if a prior attempt failed and we're now retrying. Matching
|
|
89
|
+
* both means a retried success lands in the same spot a first-try success
|
|
90
|
+
* would, preserving the surrounding capture body (the `![[memo]]` embed,
|
|
91
|
+
* the `_Recorded …_` line, the header).
|
|
92
|
+
*
|
|
93
|
+
* Deliberately NO `/g` flag — `.replace` swaps only the FIRST match. A
|
|
94
|
+
* canonical capture body holds exactly one marker, so first-match is the
|
|
95
|
+
* correct target. `applyFailureMarker`'s includes-guard (no-op when the
|
|
96
|
+
* marker is already present) prevents markers accumulating across repeated
|
|
97
|
+
* terminal failures, so the body never carries two of the same marker. A
|
|
98
|
+
* hand-edited body that somehow contains both markers patches only the
|
|
99
|
+
* first — accepted (degenerate, operator-induced).
|
|
113
100
|
*/
|
|
114
|
-
|
|
115
|
-
const raw = meta.segment_index;
|
|
116
|
-
return typeof raw === "number" && Number.isInteger(raw) && raw >= 0 ? raw : undefined;
|
|
117
|
-
}
|
|
101
|
+
const TRANSCRIPT_SUCCESS_TARGET = /_Transcript pending\._|_Transcription unavailable\._/;
|
|
118
102
|
|
|
119
103
|
/**
|
|
120
104
|
* Default sweep cadence (ms). The sweep is the safety net for backoff-
|
|
@@ -211,14 +195,6 @@ interface PendingMeta {
|
|
|
211
195
|
* worker preserves the original stub-patching behavior (Lens flow).
|
|
212
196
|
*/
|
|
213
197
|
transcribe_origin?: "auto" | "legacy";
|
|
214
|
-
/**
|
|
215
|
-
* Voice W2 (segmented recordings): a client-set 0-based index marking this
|
|
216
|
-
* attachment as one segment of a longer recording sliced into ~10-min parts,
|
|
217
|
-
* all linked on ONE note. When present, the legacy in-body path targets this
|
|
218
|
-
* part's markers (`… (part N)._`, N = segment_index + 1) rather than the bare
|
|
219
|
-
* ones — making per-part ordering structurally guaranteed. See `markersFor`.
|
|
220
|
-
*/
|
|
221
|
-
segment_index?: number;
|
|
222
198
|
[k: string]: unknown;
|
|
223
199
|
}
|
|
224
200
|
|
|
@@ -381,28 +357,17 @@ export function startTranscriptionWorker(opts: TranscriptionWorkerOpts): Transcr
|
|
|
381
357
|
* attachment failure we're trying to record.
|
|
382
358
|
*
|
|
383
359
|
* Body policy (finding F — never destroy content):
|
|
384
|
-
* -
|
|
385
|
-
* the
|
|
386
|
-
*
|
|
387
|
-
*
|
|
388
|
-
* -
|
|
389
|
-
*
|
|
390
|
-
*
|
|
391
|
-
*
|
|
392
|
-
*
|
|
393
|
-
* user's edits. We append instead so nothing is lost. If the content is
|
|
394
|
-
* empty, the marker alone becomes the body (avoids a leading blank line).
|
|
360
|
+
* - Placeholder PRESENT → surgical replace of `_Transcript pending._`
|
|
361
|
+
* with the marker. The `![[memo]]` embed + any surrounding text survive.
|
|
362
|
+
* - Marker ALREADY PRESENT → no-op (idempotent; a double-terminal-failure
|
|
363
|
+
* must not stack markers).
|
|
364
|
+
* - Otherwise (placeholder absent — the user edited the note while it was
|
|
365
|
+
* pending) → APPEND `\n\n` + marker to the existing content. The old
|
|
366
|
+
* code full-replaced the body here, destroying the embed AND the user's
|
|
367
|
+
* edits. We append instead so nothing is lost. If the content is empty,
|
|
368
|
+
* the marker alone becomes the body (avoids a leading blank line).
|
|
395
369
|
*/
|
|
396
|
-
async function applyFailureMarker(
|
|
397
|
-
store: Store,
|
|
398
|
-
noteId: string,
|
|
399
|
-
segmentIndex: number | undefined,
|
|
400
|
-
): Promise<void> {
|
|
401
|
-
// Bare markers by default; this segment's `(part N)` markers when the
|
|
402
|
-
// attachment carries a `segment_index` (voice W2). String-search replace
|
|
403
|
-
// targets the FIRST occurrence (a canonical body holds exactly one), and
|
|
404
|
-
// the includes-guard below keeps a repeated terminal failure from stacking.
|
|
405
|
-
const { pending, unavailable } = markersFor(segmentIndex);
|
|
370
|
+
async function applyFailureMarker(store: Store, noteId: string): Promise<void> {
|
|
406
371
|
// OC-guarded (vault#435): the read-transform-write below is re-run against
|
|
407
372
|
// fresh content on a conflict so a concurrent user edit isn't clobbered.
|
|
408
373
|
// The transform is pure w.r.t. the note it's handed; the stub-set and
|
|
@@ -416,24 +381,17 @@ export function startTranscriptionWorker(opts: TranscriptionWorkerOpts): Transcr
|
|
|
416
381
|
if (noteMeta.transcribe_stub !== true) return null;
|
|
417
382
|
|
|
418
383
|
let body: string;
|
|
419
|
-
if (note.content
|
|
420
|
-
|
|
421
|
-
|
|
422
|
-
body = note.content.replace(pending, () => unavailable);
|
|
423
|
-
} else if (note.content.includes(unavailable)) {
|
|
384
|
+
if (TRANSCRIPT_PLACEHOLDER.test(note.content)) {
|
|
385
|
+
body = note.content.replace(TRANSCRIPT_PLACEHOLDER, TRANSCRIPT_UNAVAILABLE);
|
|
386
|
+
} else if (note.content.includes(TRANSCRIPT_UNAVAILABLE)) {
|
|
424
387
|
// Marker already present — nothing to do. Clear the stub and
|
|
425
388
|
// return without rewriting the body so we don't stack markers.
|
|
426
389
|
body = note.content;
|
|
427
390
|
} else {
|
|
428
391
|
body = note.content.length > 0
|
|
429
|
-
? `${note.content}\n\n${
|
|
430
|
-
:
|
|
392
|
+
? `${note.content}\n\n${TRANSCRIPT_UNAVAILABLE}`
|
|
393
|
+
: TRANSCRIPT_UNAVAILABLE;
|
|
431
394
|
}
|
|
432
|
-
// Segmented: the stub is SHARED across this note's parts — keep it set
|
|
433
|
-
// so sibling parts still resolve their own slots. Return content only
|
|
434
|
-
// (leave note metadata untouched). Un-segmented: clear the one-shot
|
|
435
|
-
// stub as before (byte-unchanged).
|
|
436
|
-
if (segmentIndex !== undefined) return { content: body };
|
|
437
395
|
const { transcribe_stub: _drop, ...restMeta } = noteMeta;
|
|
438
396
|
return { content: body, metadata: restMeta };
|
|
439
397
|
},
|
|
@@ -491,12 +449,6 @@ export function startTranscriptionWorker(opts: TranscriptionWorkerOpts): Transcr
|
|
|
491
449
|
// vs. the legacy stub-patching path (Lens flow). Auto-write notes also
|
|
492
450
|
// surface failures so the user can retry from the transcript note.
|
|
493
451
|
const isAutoOrigin = meta.transcribe_origin === "auto";
|
|
494
|
-
// Voice W2: when this attachment is one segment of a longer recording
|
|
495
|
-
// (client-set `segment_index`), the legacy in-body path targets this
|
|
496
|
-
// part's markers instead of the bare ones. Undefined for un-segmented
|
|
497
|
-
// attachments — byte-unchanged behavior. Only the legacy path consults it;
|
|
498
|
-
// the auto/transcript-note path is untouched (segments are a memo concern).
|
|
499
|
-
const segmentIndex = segmentIndexOf(meta);
|
|
500
452
|
|
|
501
453
|
// Honor backoff — we re-check here in case another tick queued this
|
|
502
454
|
// attachment between the listing and now.
|
|
@@ -517,7 +469,7 @@ export function startTranscriptionWorker(opts: TranscriptionWorkerOpts): Transcr
|
|
|
517
469
|
if (isAutoOrigin) {
|
|
518
470
|
await writeFailureTranscriptNote(store, attachment, "audio file not found", undefined, undefined);
|
|
519
471
|
} else {
|
|
520
|
-
await applyFailureMarker(store, attachment.noteId
|
|
472
|
+
await applyFailureMarker(store, attachment.noteId);
|
|
521
473
|
}
|
|
522
474
|
return;
|
|
523
475
|
}
|
|
@@ -575,7 +527,7 @@ export function startTranscriptionWorker(opts: TranscriptionWorkerOpts): Transcr
|
|
|
575
527
|
if (isAutoOrigin) {
|
|
576
528
|
await writeFailureTranscriptNote(store, attachment, errMsg, apiErr?.code, undefined);
|
|
577
529
|
} else {
|
|
578
|
-
await applyFailureMarker(store, attachment.noteId
|
|
530
|
+
await applyFailureMarker(store, attachment.noteId);
|
|
579
531
|
}
|
|
580
532
|
// retention=never drops the audio on any terminal state, including
|
|
581
533
|
// failure. The user opted in to "I don't want the audio kept around
|
|
@@ -626,16 +578,6 @@ export function startTranscriptionWorker(opts: TranscriptionWorkerOpts): Transcr
|
|
|
626
578
|
// before the transcript arrives opts out of the overwrite. OC-guarded
|
|
627
579
|
// (vault#435): re-applied against fresh content on a conflict so a
|
|
628
580
|
// concurrent user edit isn't clobbered.
|
|
629
|
-
//
|
|
630
|
-
// Success replaces whichever of THIS part's markers is present (bare, or
|
|
631
|
-
// `(part N)` when segmented). Built with no `/g` flag so `.replace`
|
|
632
|
-
// swaps only the FIRST match — a canonical capture body holds exactly
|
|
633
|
-
// one marker per part; alternation preserves positional-first semantics
|
|
634
|
-
// (a retried success replaces the failure marker where a first-try
|
|
635
|
-
// success replaced the pending one) so byte-for-byte matching today's
|
|
636
|
-
// un-segmented behavior.
|
|
637
|
-
const { pending, unavailable } = markersFor(segmentIndex);
|
|
638
|
-
const successTarget = new RegExp(`${escapeRegExp(pending)}|${escapeRegExp(unavailable)}`);
|
|
639
581
|
await applyNoteTransformWithOC(
|
|
640
582
|
store,
|
|
641
583
|
attachment.noteId,
|
|
@@ -644,31 +586,28 @@ export function startTranscriptionWorker(opts: TranscriptionWorkerOpts): Transcr
|
|
|
644
586
|
const noteMeta = (note.metadata as Record<string, unknown> | undefined) ?? {};
|
|
645
587
|
if (noteMeta.transcribe_stub !== true) return null;
|
|
646
588
|
// Body policy (finding F — never destroy content):
|
|
647
|
-
// -
|
|
648
|
-
//
|
|
589
|
+
// - placeholder OR failure-marker present → surgical replace in
|
|
590
|
+
// place (a retried success replaces the `_Transcription
|
|
591
|
+
// unavailable._` marker, landing exactly where a first-try
|
|
592
|
+
// success would). The embed + surrounding capture body survive.
|
|
649
593
|
// - neither present (user edited the note while pending) → APPEND
|
|
650
594
|
// the transcript instead of full-replacing the body, so the
|
|
651
595
|
// user's edits + the `![[memo]]` embed are preserved. The old
|
|
652
596
|
// code full-replaced here, which destroyed both.
|
|
653
597
|
let body: string;
|
|
654
|
-
if (
|
|
598
|
+
if (TRANSCRIPT_SUCCESS_TARGET.test(note.content)) {
|
|
655
599
|
// Function replacer, NOT a string — speech-to-text is arbitrary
|
|
656
600
|
// user content, and String.replace treats `$&`, `$\``, `$'`,
|
|
657
601
|
// `$1`-`$9` as special patterns in a string replacement. A
|
|
658
602
|
// transcript containing `$&` would otherwise inject the matched
|
|
659
603
|
// marker text into the body. `() => transcript` returns the text
|
|
660
604
|
// verbatim.
|
|
661
|
-
body = note.content.replace(
|
|
605
|
+
body = note.content.replace(TRANSCRIPT_SUCCESS_TARGET, () => transcript);
|
|
662
606
|
} else {
|
|
663
607
|
body = note.content.length > 0
|
|
664
608
|
? `${note.content}\n\n${transcript}`
|
|
665
609
|
: transcript;
|
|
666
610
|
}
|
|
667
|
-
// Segmented: the stub is SHARED across this note's parts — keep it
|
|
668
|
-
// set so sibling parts still resolve their own slots. Return content
|
|
669
|
-
// only (leave note metadata untouched). Un-segmented: clear the
|
|
670
|
-
// one-shot stub as before (byte-unchanged).
|
|
671
|
-
if (segmentIndex !== undefined) return { content: body };
|
|
672
611
|
const { transcribe_stub: _drop, ...restMeta } = noteMeta;
|
|
673
612
|
return { content: body, metadata: restMeta };
|
|
674
613
|
},
|