@vellumai/assistant 0.12.0-dev.202609111819.3dcbf21 → 0.12.0-dev.202609112015.230c060

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (33) hide show
  1. package/ARCHITECTURE.md +1 -1
  2. package/docs/architecture/turn-actor.md +15 -8
  3. package/openapi.yaml +12 -0
  4. package/package.json +1 -1
  5. package/src/__tests__/always-loaded-tools-guard.test.ts +10 -6
  6. package/src/__tests__/conversation-process-app-control-preactivation.test.ts +4 -10
  7. package/src/__tests__/conversation-queue.test.ts +48 -207
  8. package/src/__tests__/conversation-surfaces-task-progress.test.ts +67 -0
  9. package/src/__tests__/drain-kick-guard.test.ts +0 -2
  10. package/src/__tests__/drain-requeue-on-contention.test.ts +0 -6
  11. package/src/__tests__/subagent-tool-gate-mode.test.ts +12 -3
  12. package/src/__tests__/watch-retro-tool-availability.test.ts +8 -17
  13. package/src/acp/__tests__/session-manager.test.ts +39 -11
  14. package/src/acp/session-manager.ts +8 -2
  15. package/src/calls/__tests__/barge-in-guard.test.ts +81 -0
  16. package/src/calls/__tests__/call-controller.test.ts +22 -24
  17. package/src/calls/__tests__/media-stream-server-integration.test.ts +114 -24
  18. package/src/calls/barge-in-guard.ts +118 -0
  19. package/src/calls/call-controller.ts +29 -27
  20. package/src/calls/media-stream-server.ts +82 -30
  21. package/src/calls/media-stream-stt-session.ts +15 -0
  22. package/src/config/bundled-skills/acp/SKILL.md +2 -0
  23. package/src/config/bundled-skills/acp/TOOLS.json +1 -1
  24. package/src/daemon/__tests__/conversation-tool-setup.test.ts +32 -28
  25. package/src/daemon/conversation-process.ts +35 -64
  26. package/src/daemon/conversation-surfaces.ts +7 -5
  27. package/src/daemon/conversation-tool-setup.ts +4 -8
  28. package/src/live-voice/live-voice-session.ts +26 -76
  29. package/src/runtime/routes/__tests__/acp-routes.test.ts +32 -2
  30. package/src/runtime/routes/acp-routes.ts +19 -1
  31. package/src/tools/acp/spawn.test.ts +61 -4
  32. package/src/tools/acp/spawn.ts +26 -15
  33. package/src/tools/watch/watch-retro-report.ts +0 -8
package/ARCHITECTURE.md CHANGED
@@ -609,7 +609,7 @@ Every phone call connects over Twilio Media Streams: the voice webhook emits `<C
609
609
 
610
610
  Transcription mode is selected once per session in `media-stream-stt-session.ts`:
611
611
 
612
- - **Streaming** (default): when `calls.voice.telephonyStreaming` is enabled and the `telephony` role resolves a streaming transcriber (`resolveStreamingTranscriber({ role: "telephony" })`), inbound audio is decoded (mu-law → PCM16, resampled 8 kHz → 16 kHz) and fed to the provider's realtime adapter. Replies trigger only on utterance-boundary finals (for Deepgram, `speech_final`/`UtteranceEnd`, never mid-sentence `is_final` segments), and barge-in fires from local energy VAD, never from transcriber partials.
612
+ - **Streaming** (default): when `calls.voice.telephonyStreaming` is enabled and the `telephony` role resolves a streaming transcriber (`resolveStreamingTranscriber({ role: "telephony" })`), inbound audio is decoded (mu-law → PCM16, resampled 8 kHz → 16 kHz) and fed to the provider's realtime adapter. Replies trigger only on utterance-boundary finals (for Deepgram, `speech_final`/`UtteranceEnd`, never mid-sentence `is_final` segments), and barge-in fires from local energy VAD, never from transcriber partials. While something is interruptible (an assistant turn in flight, thinking or speaking, or a completed turn's tail still playing from Twilio's buffer), caller speech arms the shared sustained-speech barge-in guard (`src/calls/barge-in-guard.ts`, the same accounting live voice uses: speech accumulates toward 250 ms, short gaps are tolerated, a run that is mostly silence resets) and every inbound frame feeds it; outside that window the guard is dropped, so the caller's own utterance never carries into a turn that starts before the local VAD ends it. Only a fired guard reaches `CallController.handleBargeIn`, which interrupts a turn in either phase and ignores an idle controller (the playing tail is cleared instead).
613
613
  - **Batch fallback**: otherwise the session segments turns with the energy-based `MediaTurnDetector` and transcribes each completed turn via the same role's batch API. Both halves of a call read the `telephony` role, which is why a role names its consumer rather than a boundary.
614
614
 
615
615
  Every phone turn runs the same two-leg triage as live voice through `startVoiceTurn` (`src/calls/voice-session-bridge.ts`): `call-controller.ts` opens on a toolless front-door leg (`routingLeg: "front-door"`, the `voiceFrontDoor` call site) and drives it through the shared `createFrontDoorLegCoordinator` (`src/calls/voice-leg-coordinator.ts`), which reads the stream through the verdict machine and sequences the hand-off (pause narration, abort the leg, resolve and speak the bridge, mark it as the floor holder, start the escalated leg pinned to the conversation's own model, re-arm narration); each driver supplies only a host for how text and the bridge are spoken, how a leg is started or aborted, and (live voice only) the speculative hold and commit. Phone has no partial transcripts, so the hold verdict is never taught and routing is escalate-only. The controller also passes the bridge's turn callbacks (tool activity is recorded as `tool_use_started` / `tool_use_completed` call events, persisted row ids ride the `assistant_spoke` event), `launchedAtMs` for dispatch timing, and a `voiceTelemetry` bag keyed by the call session with a `phone_inbound` / `phone_outbound` entry. Both drivers share the spoken progress narration cadence (`src/calls/voice-progress-cadence.ts`, tuned by `voice.frontModel.progress`): the cadence owns the tool-activity log, the triggers (an ops burst, a long operation completing, a full interval of audible silence with news, the `maxSilenceMs` heartbeat) and the generated or static phrase, while each driver supplies its own view of audible silence (live voice from its TTS queue and playback-tail estimate; the media-stream transport from `isPlaybackIdle()` and a running sum of sent frame durations) and how to speak a phrase.
@@ -60,14 +60,21 @@ and restore the prior value afterwards, each guarding the restore so a turn that
60
60
  started in between is not clobbered. They are supplying the acting actor for
61
61
  their run, and are covered by this contract.
62
62
 
63
- A queued message commits to a run at its drain, not at its enqueue. The drains
64
- (`drainSingleMessage` and `drainBatch` in `conversation-process.ts`) stamp the
65
- queued sender the way `processMessage` stamps its committing actor, and re-scope
66
- the resident history to them, so a turn drained behind another actor's turn
67
- runs as its sender everywhere the resting actor is read, not only in the
68
- per-turn snapshot. A steered drain skips the re-scope: its sender owned the
69
- turn it cut off, and the resident history may carry the in-memory repair of the
70
- abandoned `tool_use`.
63
+ The queue drains are deliberately not in that set. `drainSingleMessage` and
64
+ `drainBatch` carry the queued sender on the per-turn field and into the run,
65
+ and leave the resting slot alone: at the point they stamp, the drain has not
66
+ yet proved it holds the processing lock. Its `isProcessing()` check is a
67
+ time-of-check guard whose documented backstop is the persist that can still
68
+ fail busy, and the requeue on that path restores the queue and the steer flag
69
+ only. A drain that stamped the slot would therefore leave the conversation
70
+ attributed to a sender whose turn never ran.
71
+
72
+ The consequence is that a cross-actor drain runs against history scoped for the
73
+ previous actor, because `ensureActorScopedHistory` reads the slot. That is a
74
+ real defect and it is LUM-3344's, which owns the fix: scope the transcript at
75
+ assembly time from the turn's actor, rather than mutating a shared transcript
76
+ and tracking who it was scoped for. Do not close it by making the drain stamp
77
+ the slot first.
71
78
 
72
79
  `call-controller` keeps its own `trustContext` on its own object and never reads
73
80
  the conversation's. It is outside this contract.
package/openapi.yaml CHANGED
@@ -596,6 +596,16 @@ paths:
596
596
  type: string
597
597
  agent:
598
598
  type: string
599
+ requestedModel:
600
+ anyOf:
601
+ - type: string
602
+ - type: "null"
603
+ description: The model explicitly requested for this spawn, if any.
604
+ effectiveModel:
605
+ anyOf:
606
+ - type: string
607
+ - type: "null"
608
+ description: The top-level session model reported by the ACP adapter, if any.
599
609
  modelWarning:
600
610
  description: Why the requested model was not applied. The session is running on the agent's own model.
601
611
  type: string
@@ -603,6 +613,8 @@ paths:
603
613
  - acpSessionId
604
614
  - protocolSessionId
605
615
  - agent
616
+ - requestedModel
617
+ - effectiveModel
606
618
  additionalProperties: false
607
619
  /v1/activation/dismiss:
608
620
  post:
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@vellumai/assistant",
3
- "version": "0.12.0-dev.202609111819.3dcbf21",
3
+ "version": "0.12.0-dev.202609112015.230c060",
4
4
  "license": "MIT",
5
5
  "type": "module",
6
6
  "exports": {
@@ -26,7 +26,7 @@ afterAll(() => {
26
26
  });
27
27
 
28
28
  describe("always-loaded tool count", () => {
29
- test("should be exactly 11 with recall occupying the existing slot", async () => {
29
+ test("should be exactly 14 with recall occupying the existing slot", async () => {
30
30
  await initializeTools();
31
31
  const allDefs = getAllToolDefinitions();
32
32
 
@@ -48,10 +48,11 @@ describe("always-loaded tool count", () => {
48
48
  // connected — without a human in the loop, the guardian auto-approve
49
49
  // path would allow unchecked host command execution.
50
50
  //
51
- // `watch_retro_report` is here for the same reason the ui_surface tools are
52
- // NOT: a watch retrospective runs clientless, so it can only report through
53
- // a tool that survives this baseline. Its description and schema are kept
54
- // deliberately terse because that is the cost of the slot.
51
+ // `watch_retro_report` survives this baseline for the same reason the
52
+ // ui_* tools do: a watch retrospective runs clientless, so it can only
53
+ // report through a tool that survives this baseline. The ui_* tools are
54
+ // here because background UI surfaces persist and return instead of
55
+ // awaiting action, so they no longer need a connected client.
55
56
  const expectedNames = [
56
57
  "bash",
57
58
  "file_edit",
@@ -61,6 +62,9 @@ describe("always-loaded tool count", () => {
61
62
  "remember",
62
63
  "skill_execute",
63
64
  "skill_load",
65
+ "ui_dismiss",
66
+ "ui_show",
67
+ "ui_update",
64
68
  "watch_retro_report",
65
69
  "web_fetch",
66
70
  "web_search",
@@ -68,6 +72,6 @@ describe("always-loaded tool count", () => {
68
72
 
69
73
  expect(activeNames).toEqual(expectedNames);
70
74
  expect(activeNames.filter((name) => name === "recall")).toHaveLength(1);
71
- expect(activeTools.length).toBe(11);
75
+ expect(activeTools.length).toBe(14);
72
76
  });
73
77
  });
@@ -161,12 +161,6 @@ function makeFakeContext(opts: {
161
161
  trustClass: "guardian" as const,
162
162
  guardianPrincipalId: "user-1",
163
163
  },
164
- setTrustContext(
165
- this: { trustContext?: TrustContext },
166
- trustContext: TrustContext | null,
167
- ) {
168
- this.trustContext = trustContext ?? undefined;
169
- },
170
164
  setTransportHints() {},
171
165
  applyHostEnvFromTransport() {},
172
166
  ensureHostProxiesForTurn() {},
@@ -415,10 +409,10 @@ describe("drainQueue preactivation re-add for host-proxy interfaces", () => {
415
409
  "U-contact",
416
410
  );
417
411
  expect(ctx.currentTurnTrustContext?.sourceChannel).toBe("slack");
418
- // The drain is where a queued message commits to a run, so the resting
419
- // slot names the sender too: history is scoped from it, and a slot still
420
- // naming the guardian would hand the contact's turn the guardian's rows.
421
- expect(ctx.trustContext).toBe(contactTrust);
412
+ // The slot itself is left alone; only the turn's view is corrected. A
413
+ // drain stamping it would attribute the conversation to a sender whose
414
+ // turn can still lose the processing lock at persist time.
415
+ expect(ctx.trustContext?.trustClass).toBe("guardian");
422
416
  });
423
417
 
424
418
  test("buildPassthroughBatch refuses to coalesce two channel senders", async () => {
@@ -450,17 +450,6 @@ async function waitForCondition(
450
450
  }
451
451
  }
452
452
 
453
- /** Count the history reloads a conversation performs from this point on. */
454
- function countHistoryReloads(conversation: Conversation): () => number {
455
- const originalLoad = conversation.loadFromDb.bind(conversation);
456
- let reloads = 0;
457
- conversation.loadFromDb = async () => {
458
- reloads++;
459
- return originalLoad();
460
- };
461
- return () => reloads;
462
- }
463
-
464
453
  /**
465
454
  * Resolve the Nth pending AgentLoop.run() call. Fires the minimal events
466
455
  * that `runAgentLoop` expects (usage + message_complete) so the conversation
@@ -726,22 +715,21 @@ describe("Conversation message queue", () => {
726
715
  await new Promise((r) => setTimeout(r, 10));
727
716
  });
728
717
 
729
- test("a drained turn commits its sender as the resting actor and re-scopes history", async () => {
730
- // A contact's channel turn leaves the resting slot, and the resident
731
- // history scoped from it, naming the contact. The guardian's message
732
- // queued behind that turn must run as the guardian everywhere the slot is
733
- // read, not only in the per-turn snapshot: the `<turn_context>` actor
734
- // section is derived at turn start and history is scoped from the slot,
735
- // so a drain that stamped only the snapshot rendered the guardian's own
736
- // web turn as a trusted contact's and hid the guardian's rows from it.
718
+ test("the turn-context actor section describes the turn's actor, not the conversation's resting actor", async () => {
719
+ // The drain carries its sender on the per-turn field and leaves the
720
+ // resting slot naming the previous actor. The actor section is frozen at
721
+ // turn start, so it has to read the turn's actor: reading the slot is what
722
+ // told the guardian's own drained turn it was talking to a trusted
723
+ // contact.
737
724
  const conversation = makeConversation();
738
725
  await conversation.loadFromDb();
739
726
 
740
- conversation.setTrustContext({
741
- trustClass: "trusted_contact",
742
- sourceChannel: "slack",
727
+ const contact = {
728
+ trustClass: "trusted_contact" as const,
729
+ sourceChannel: "slack" as const,
743
730
  requesterExternalUserId: "U-contact",
744
- });
731
+ };
732
+ conversation.setTrustContext(contact);
745
733
 
746
734
  const p1 = conversation.processMessage({
747
735
  content: "msg-1",
@@ -756,7 +744,6 @@ describe("Conversation message queue", () => {
756
744
  sourceChannel: "vellum" as const,
757
745
  requesterExternalUserId: "guardian-principal",
758
746
  };
759
- const reloads = countHistoryReloads(conversation);
760
747
  conversation.enqueueMessage({
761
748
  content: "msg-2",
762
749
  requestId: "req-2",
@@ -767,83 +754,30 @@ describe("Conversation message queue", () => {
767
754
  await p1;
768
755
  await waitForPendingRun(2);
769
756
 
770
- // Read while run 2 is in flight: the slot, the snapshot, and the frozen
771
- // actor section all describe the guardian, and the history was reloaded
772
- // for that scope exactly once.
773
- expect(conversation.getTrustContext()).toBe(guardian);
757
+ // Guard the test itself: the slot must disagree with the turn, or the
758
+ // assertion below passes for the wrong reason.
759
+ expect(conversation.getTrustContext()).toBe(contact);
774
760
  expect(conversation.currentTurnTrustContext).toBe(guardian);
761
+ // A guardian turn renders no actor section, whatever the slot says.
775
762
  expect(conversation.currentTurnInboundActorContext).toBeNull();
776
- expect(reloads()).toBe(1);
777
763
 
778
764
  await resolveRun(1);
779
765
  await new Promise((r) => setTimeout(r, 10));
780
766
  });
781
767
 
782
- test("a batched drain commits the head's sender as the resting actor", async () => {
783
- // Same commitment on the batched path, which coalesces only messages that
784
- // share a trust identity, so the head's sender is the batch's actor.
768
+ test("the actor section names the drained contact while the slot holds the guardian", async () => {
769
+ // The inverse direction, so the assertion above cannot pass merely because
770
+ // the section is always absent: a contact's message drained behind the
771
+ // guardian's turn must be described as that contact.
785
772
  const conversation = makeConversation();
786
773
  await conversation.loadFromDb();
787
774
 
788
- conversation.setTrustContext({
789
- trustClass: "trusted_contact",
790
- sourceChannel: "slack",
791
- requesterExternalUserId: "U-contact",
792
- });
793
-
794
- const p1 = conversation.processMessage({
795
- content: "msg-1",
796
- attachments: [],
797
- onEvent: () => {},
798
- requestId: "req-1",
799
- });
800
- await waitForPendingRun(1);
801
-
802
775
  const guardian = {
803
776
  trustClass: "guardian" as const,
804
777
  sourceChannel: "vellum" as const,
805
778
  requesterExternalUserId: "guardian-principal",
806
779
  };
807
- const reloads = countHistoryReloads(conversation);
808
- conversation.enqueueMessage({
809
- content: "msg-2",
810
- requestId: "req-2",
811
- trustContext: guardian,
812
- });
813
- conversation.enqueueMessage({
814
- content: "msg-3",
815
- requestId: "req-3",
816
- trustContext: guardian,
817
- });
818
-
819
- await resolveRun(0);
820
- await p1;
821
- await waitForPendingRun(2);
822
-
823
- // One batched run for both siblings, committed as the guardian.
824
- expect(pendingRuns.length).toBe(2);
825
- expect(conversation.getTrustContext()).toBe(guardian);
826
- expect(conversation.currentTurnTrustContext).toBe(guardian);
827
- expect(conversation.currentTurnInboundActorContext).toBeNull();
828
- expect(reloads()).toBe(1);
829
-
830
- await resolveRun(1);
831
- await new Promise((r) => setTimeout(r, 10));
832
- });
833
-
834
- test("a drained turn keeps its sender when the slot moves during the history reload", async () => {
835
- // The commit captures the sender before the reload awaits. A writer that
836
- // moves the slot inside that await (a wake's stamp, a pointer elevation)
837
- // must not become the turn's actor: the history was reloaded for the
838
- // sender, and running someone else's trust over it is the escalation.
839
- const conversation = makeConversation();
840
- await conversation.loadFromDb();
841
-
842
- conversation.setTrustContext({
843
- trustClass: "trusted_contact",
844
- sourceChannel: "slack",
845
- requesterExternalUserId: "U-contact",
846
- });
780
+ conversation.setTrustContext(guardian);
847
781
 
848
782
  const p1 = conversation.processMessage({
849
783
  content: "msg-1",
@@ -853,49 +787,39 @@ describe("Conversation message queue", () => {
853
787
  });
854
788
  await waitForPendingRun(1);
855
789
 
856
- const guardian = {
857
- trustClass: "guardian" as const,
858
- sourceChannel: "vellum" as const,
859
- requesterExternalUserId: "guardian-principal",
860
- };
861
- const intruder = {
862
- trustClass: "unknown" as const,
863
- sourceChannel: "telegram" as const,
864
- requesterExternalUserId: "T-stranger",
790
+ const contact = {
791
+ trustClass: "trusted_contact" as const,
792
+ sourceChannel: "slack" as const,
793
+ requesterExternalUserId: "U-contact",
865
794
  };
866
795
  conversation.enqueueMessage({
867
796
  content: "msg-2",
868
797
  requestId: "req-2",
869
- trustContext: guardian,
798
+ trustContext: contact,
870
799
  });
871
800
 
872
- // The drain's reload is the await the writer lands inside.
873
- const originalLoad = conversation.loadFromDb.bind(conversation);
874
- let movedDuringReload = false;
875
- conversation.loadFromDb = async () => {
876
- const result = await originalLoad();
877
- conversation.setTrustContext(intruder);
878
- movedDuringReload = true;
879
- return result;
880
- };
881
-
882
801
  await resolveRun(0);
883
802
  await p1;
884
803
  await waitForPendingRun(2);
885
804
 
886
- expect(movedDuringReload).toBe(true);
887
- expect(conversation.getTrustContext()).toBe(intruder);
888
- expect(conversation.currentTurnTrustContext).toBe(guardian);
889
- expect(conversation.currentTurnInboundActorContext).toBeNull();
805
+ expect(conversation.getTrustContext()).toBe(guardian);
806
+ expect(conversation.currentTurnTrustContext).toBe(contact);
807
+ expect(conversation.currentTurnInboundActorContext).toMatchObject({
808
+ trustClass: "trusted_contact",
809
+ sourceChannel: "slack",
810
+ canonicalActorIdentity: "U-contact",
811
+ });
890
812
 
891
813
  await resolveRun(1);
892
814
  await new Promise((r) => setTimeout(r, 10));
893
815
  });
894
816
 
895
- test("a drain whose history reload fails puts the resting actor back", async () => {
896
- // A reload that fails starts no turn: the message is requeued for the
897
- // next drain, so the slot must not keep naming a sender whose turn never
898
- // began, or conversation-level readers report an owner that is not there.
817
+ test("a processMessage turn whose history reload fails puts the resting actor back", async () => {
818
+ // `processMessage` is the commitment point, so it stamps the slot before
819
+ // scoping history. A reload that throws starts no turn, and leaving the
820
+ // stamp would attribute the conversation to a sender that never ran: an
821
+ // actorless dispatch afterwards (a deferred wake) resolves the resting
822
+ // slot and would inherit it.
899
823
  const conversation = makeConversation();
900
824
  await conversation.loadFromDb();
901
825
 
@@ -906,110 +830,27 @@ describe("Conversation message queue", () => {
906
830
  };
907
831
  conversation.setTrustContext(contact);
908
832
 
909
- const p1 = conversation.processMessage({
910
- content: "msg-1",
911
- attachments: [],
912
- onEvent: () => {},
913
- requestId: "req-1",
914
- });
915
- await waitForPendingRun(1);
916
-
917
- const guardian = {
918
- trustClass: "guardian" as const,
919
- sourceChannel: "vellum" as const,
920
- requesterExternalUserId: "guardian-principal",
921
- };
922
- conversation.enqueueMessage({
923
- content: "msg-2",
924
- requestId: "req-2",
925
- trustContext: guardian,
926
- });
927
-
928
- let reloadAttempts = 0;
929
833
  conversation.loadFromDb = async () => {
930
- reloadAttempts++;
931
834
  throw new Error("history store exploded");
932
835
  };
933
836
 
934
- // Finish the first turn; the drain (and its one retry) fail on the reload.
935
- await resolveRun(0);
936
- await p1;
937
- await waitForCondition(() => reloadAttempts >= 2);
938
- await new Promise((r) => setTimeout(r, 20));
939
-
940
- expect(pendingRuns.length).toBe(1);
941
- expect(conversation.getQueueDepth()).toBe(1);
942
- expect(conversation.getTrustContext()).toBe(contact);
943
- });
944
-
945
- test("the turn-context actor section follows the turn's actor when the slot moves before the loop opens", async () => {
946
- // The actor section is frozen at turn start from the turn's own actor,
947
- // not from the resting slot. The drain stamps the slot and then awaits
948
- // (persist) before the loop opens; an out-of-band writer landing in that
949
- // window (pointer elevation restore, a wake's restore) moves the slot
950
- // without owning the turn, and a read of the slot there would describe
951
- // that writer's actor to the model.
952
- const conversation = makeConversation();
953
- await conversation.loadFromDb();
954
-
955
- const contact = {
956
- trustClass: "trusted_contact" as const,
957
- sourceChannel: "slack" as const,
958
- requesterExternalUserId: "U-contact",
959
- };
960
- conversation.setTrustContext(contact);
961
-
962
- const p1 = conversation.processMessage({
963
- content: "msg-1",
964
- attachments: [],
965
- onEvent: () => {},
966
- requestId: "req-1",
967
- });
968
- await waitForPendingRun(1);
969
-
970
837
  const guardian = {
971
838
  trustClass: "guardian" as const,
972
839
  sourceChannel: "vellum" as const,
973
840
  requesterExternalUserId: "guardian-principal",
974
841
  };
975
- conversation.enqueueMessage({
976
- content: "msg-2",
977
- requestId: "req-2",
978
- trustContext: guardian,
979
- });
980
-
981
- // Land the out-of-band slot write inside the drain's window, after its
982
- // stamp and before the loop opens: the drain awaits persistUserMessage
983
- // there. The resting slot alone is moved; the per-turn snapshot is the
984
- // drain's to carry into the run.
985
- const originalPersist = conversation.persistUserMessage.bind(conversation);
986
- let slotMoved = false;
987
- (
988
- conversation as unknown as {
989
- persistUserMessage: typeof conversation.persistUserMessage;
990
- }
991
- ).persistUserMessage = async (opts) => {
992
- const result = await originalPersist(opts);
993
- conversation.setTrustContext(contact);
994
- slotMoved = true;
995
- return result;
996
- };
997
-
998
- await resolveRun(0);
999
- await p1;
1000
- await waitForPendingRun(2);
842
+ await expect(
843
+ conversation.processMessage({
844
+ content: "msg-1",
845
+ attachments: [],
846
+ onEvent: () => {},
847
+ requestId: "req-1",
848
+ trustContext: guardian,
849
+ }),
850
+ ).rejects.toThrow("history store exploded");
1001
851
 
1002
- // Guard the test itself: the injection must have run, and the slot must
1003
- // disagree with the turn, or the assertion below passes for the wrong
1004
- // reason.
1005
- expect(slotMoved).toBe(true);
1006
852
  expect(conversation.getTrustContext()).toBe(contact);
1007
- expect(conversation.currentTurnTrustContext).toBe(guardian);
1008
- // The guardian's turn renders no actor section, whatever the slot says.
1009
- expect(conversation.currentTurnInboundActorContext).toBeNull();
1010
-
1011
- await resolveRun(1);
1012
- await new Promise((r) => setTimeout(r, 10));
853
+ expect(pendingRuns.length).toBe(0);
1013
854
  });
1014
855
 
1015
856
  test("a processMessage turn keeps its turn-start trust when the slot moves before the loop opens", async () => {
@@ -21,6 +21,7 @@ import {
21
21
  function makeContext(
22
22
  sent: AssistantEvent[] = [],
23
23
  channelCapabilities?: { channel: string; supportsDynamicUi: boolean },
24
+ opts?: { hasNoClient?: boolean },
24
25
  ): Conversation {
25
26
  return asConversation({
26
27
  conversationId: "session-1",
@@ -37,6 +38,7 @@ function makeContext(
37
38
  accumulatedSurfaceState: new Map<string, Record<string, unknown>>(),
38
39
  surfaceActionRequestIds: new Set<string>(),
39
40
  currentTurnSurfaces: [],
41
+ hasNoClient: opts?.hasNoClient ?? false,
40
42
  isProcessing: () => false,
41
43
  enqueueMessage: () => ({ queued: false, requestId: "req-1" }),
42
44
  getQueueDepth: () => 0,
@@ -65,6 +67,71 @@ describe("task_progress surface compatibility", () => {
65
67
  expect(sent).toHaveLength(0);
66
68
  });
67
69
 
70
+ test("persists a clientless choice from a non-rendering channel without waiting", async () => {
71
+ const sent: AssistantEvent[] = [];
72
+ const ctx = makeContext(
73
+ sent,
74
+ {
75
+ channel: "phone",
76
+ supportsDynamicUi: false,
77
+ },
78
+ { hasNoClient: true },
79
+ );
80
+
81
+ const result = await surfaceProxyResolver(ctx, "ui_show", {
82
+ surface_type: "choice",
83
+ title: "Choose a focus",
84
+ data: {
85
+ options: [
86
+ { id: "inbox", title: "Inbox" },
87
+ { id: "calendar", title: "Calendar" },
88
+ ],
89
+ },
90
+ });
91
+
92
+ expect(result.isError).toBe(false);
93
+ expect(result.yieldToUser).toBeUndefined();
94
+ const { surfaceId } = JSON.parse(result.content) as { surfaceId: string };
95
+ expect(ctx.currentTurnSurfaces.some((s) => s.surfaceId === surfaceId)).toBe(
96
+ true,
97
+ );
98
+ expect(ctx.pendingSurfaceActions.has(surfaceId)).toBe(false);
99
+ expect(sent.some((msg) => msg.type === "ui_surface_show")).toBe(true);
100
+ });
101
+
102
+ test("persists a clientless update in the current-turn surface snapshot", async () => {
103
+ const sent: AssistantEvent[] = [];
104
+ const ctx = makeContext(
105
+ sent,
106
+ {
107
+ channel: "phone",
108
+ supportsDynamicUi: false,
109
+ },
110
+ { hasNoClient: true },
111
+ );
112
+ const shown = await surfaceProxyResolver(ctx, "ui_show", {
113
+ surface_type: "card",
114
+ title: "Background work",
115
+ data: {
116
+ template: "task_progress",
117
+ templateData: { status: "in_progress", steps: [] },
118
+ },
119
+ });
120
+ const { surfaceId } = JSON.parse(shown.content) as { surfaceId: string };
121
+
122
+ const result = await surfaceProxyResolver(ctx, "ui_update", {
123
+ surface_id: surfaceId,
124
+ data: { templateData: { status: "completed" } },
125
+ });
126
+
127
+ expect(result.isError).toBe(false);
128
+ const data = ctx.currentTurnSurfaces.find((s) => s.surfaceId === surfaceId)
129
+ ?.data as CardSurfaceData;
130
+ expect((data.templateData as Record<string, unknown>).status).toBe(
131
+ "completed",
132
+ );
133
+ });
134
+
68
135
  test("blocks ui_update when channel lacks dynamic UI support", async () => {
69
136
  const sent: AssistantEvent[] = [];
70
137
  const ctx = makeContext(sent, {
@@ -79,8 +79,6 @@ function makeFakeConversation(
79
79
  getTurnInterfaceContext: () => null,
80
80
  setTurnInterfaceContext: () => {},
81
81
  setTransportHints: () => {},
82
- setTrustContext: () => {},
83
- ensureActorScopedHistory: async () => {},
84
82
  emitActivityState: () => {
85
83
  if (activityFailures > 0) {
86
84
  activityFailures -= 1;
@@ -78,12 +78,6 @@ function makeFakeConversation(options: {
78
78
  setTransportHints: () => {
79
79
  mutationCalls.push("setTransportHints");
80
80
  },
81
- setTrustContext: () => {
82
- mutationCalls.push("setTrustContext");
83
- },
84
- ensureActorScopedHistory: async () => {
85
- mutationCalls.push("ensureActorScopedHistory");
86
- },
87
81
  emitActivityState: () => {
88
82
  mutationCalls.push("emitActivityState");
89
83
  },
@@ -249,14 +249,19 @@ describe("createResolveToolsCallback — toolContextPin", () => {
249
249
  });
250
250
  }
251
251
 
252
- test("control: without a pin, a clientless fork drops every client-gated tool from the wire", () => {
252
+ test("control: without a pin, a clientless fork drops every client-gated tool but ui_show", () => {
253
253
  projectedSkillToolNames = [];
254
254
  const resolve = createResolveToolsCallback(
255
255
  CLIENT_GATED_DEFS,
256
256
  clientlessExecutionCtx(),
257
257
  )!;
258
258
 
259
- expect(resolve(EMPTY_HISTORY).map((t) => t.name)).toEqual(["remember"]);
259
+ // ui_show stays on the wire: background UI surfaces persist and return
260
+ // instead of awaiting action, so they no longer need a connected client.
261
+ expect(resolve(EMPTY_HISTORY).map((t) => t.name)).toEqual([
262
+ "remember",
263
+ "ui_show",
264
+ ]);
260
265
  });
261
266
 
262
267
  test("a desktop-source pin restores the host/UI/client tool defs on the wire", () => {
@@ -291,7 +296,11 @@ describe("createResolveToolsCallback — toolContextPin", () => {
291
296
  }),
292
297
  )!;
293
298
 
294
- expect(resolve(EMPTY_HISTORY).map((t) => t.name)).toEqual(["remember"]);
299
+ // ui_show survives the clientless pin: it persists and returns.
300
+ expect(resolve(EMPTY_HISTORY).map((t) => t.name)).toEqual([
301
+ "remember",
302
+ "ui_show",
303
+ ]);
295
304
  });
296
305
 
297
306
  test("invariant: a pinned-in tool is on the wire but can never execute", async () => {