@vellumai/assistant 0.11.4-staging.3 → 0.11.4

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -54,6 +54,41 @@ anti-pattern was retired in #35642 and again in the ask_question redesign).
54
54
  | Kind-specific follow-through | resolver registry (`kind` → resolver) | switch statements in the router |
55
55
  | Reply understanding | guardian reply router (codes, buttons, reactions, modes) | per-feature inbound intercepts |
56
56
 
57
+ ## Cards are not conversation history
58
+
59
+ `pairDeliveryWithConversation` persists one message row per delivery so the
60
+ card renders and deep-links. For a guardian card that row is addressed to a
61
+ conversation the request is _about_, not one the assistant is speaking in:
62
+ `buildVellumCardAffinity` pins the vellum card to the originating
63
+ conversation, and a channel card lands in whatever conversation the guardian's
64
+ chat binds to. Either way the row is written straight to the DB by the
65
+ notification pipeline, so the live turn's in-memory history never sees it.
66
+
67
+ That row must never be replayed to the model. The conversation it lands in is
68
+ typically parked mid-approval, with its last assistant message carrying the
69
+ `tool_use` still waiting on this very decision. Replayed, the card sits
70
+ between that `tool_use` and its `tool_result`; history repair reads the pair
71
+ as broken, synthesizes a stub result, and downgrades the real one to text. A
72
+ card left as the tail row instead ends the history on an assistant message,
73
+ which extended-thinking models reject outright ("does not support assistant
74
+ message prefill").
75
+
76
+ `isGuardianCardRow` (`notifications/approval-card-data.ts`) is the one
77
+ definition of which rows those are, read off the card's own `ui_surface` id
78
+ using the same prefixes `approvalCardSurfaceId` recomputes for withdrawal. It
79
+ is derived rather than stored so a row written before the rule existed is
80
+ recognized on the same terms as a new one, with no marker to backfill.
81
+
82
+ **Both history assemblers must consult it.** `Conversation.loadFromDb` builds
83
+ `this.messages`, but Slack conversations do not use that list:
84
+ `loadSlackChronologicalContext` re-reads the rows. A rule applied to only one
85
+ exempts the other channel. Each applies it _after_ its own compaction boundary,
86
+ since both boundaries are computed against unfiltered row lists.
87
+
88
+ Surface state is deliberately NOT filtered this way. `restoreSurfaceStateFromHistory`
89
+ takes the pre-filter window, because a card's Approve/Reject buttons must keep
90
+ routing after a restart even though the card is absent from the model's history.
91
+
57
92
  Two instruction modes exist per request kind (`notifications/guardian-question-mode.ts`):
58
93
  **approval** ("CODE approve" / approve–reject buttons) and **answer**
59
94
  ("CODE <your answer>" / option buttons). `pending_question` is answer-mode.
@@ -11,6 +11,12 @@
11
11
  * (`gateway/src/http/routes/remote-web-pairing-verification.ts`)
12
12
  * - `POST /v1/remote-web/pairing-token` poll + exchange device code
13
13
  * (`gateway/src/http/routes/remote-web-pairing-token.ts`)
14
+ * - `GET /v1/remote-web/pairing-requests` list pending challenges
15
+ * (loopback-only)
16
+ * - `POST /v1/remote-web/pairing-requests/approve` approve by request id
17
+ * (loopback-only)
18
+ * - `POST /v1/remote-web/pairing-requests/deny` deny (delete) by request id
19
+ * (loopback-only)
14
20
  *
15
21
  * These shapes mirror those handlers' request/response bodies exactly so the
16
22
  * gateway, the `vellum pair` CLI (`cli/src/commands/pair.ts`), and the web SPA
@@ -68,6 +74,60 @@ export interface RemoteWebPairingVerificationResponse {
68
74
  expiresAt: string;
69
75
  }
70
76
 
77
+ /**
78
+ * One pending challenge as shown on a host approval surface.
79
+ *
80
+ * The requesting device already sees the plaintext `userCode` in its own
81
+ * challenge response ({@link RemoteWebPairingChallengeResponse.userCode});
82
+ * the loopback-gated list route is the only host-side re-exposure. Displaying
83
+ * it there is what lets the approver match the code against the requesting
84
+ * device's screen: the device-flow anti-phishing binding.
85
+ */
86
+ export interface RemoteWebPairingRequestSummary {
87
+ /** Opaque server-side id used to approve or deny this request. */
88
+ requestId: string;
89
+ /** The human-readable code the requesting device is displaying (e.g. "ABCD-EFGH"). */
90
+ userCode: string;
91
+ /** Public base URL the challenge was minted for. */
92
+ publicBaseUrl: string;
93
+ /** ISO-8601 instant the challenge was minted. */
94
+ requestedAt: string;
95
+ /** ISO-8601 instant the challenge expires. */
96
+ expiresAt: string;
97
+ /**
98
+ * Client IP of the mint request: the loopback/host address when minted
99
+ * locally, or the edge-observed client address when the mint arrived
100
+ * through the nginx tunnel edge (which stamps it via `proxy_set_header`,
101
+ * so a remote client cannot smuggle a value).
102
+ */
103
+ requesterIp: string;
104
+ /** User-Agent header of the mint request, or null when absent. */
105
+ requesterUserAgent: string | null;
106
+ /**
107
+ * Whether the mint arrived through the public tunnel edge rather than the
108
+ * host itself.
109
+ */
110
+ viaEdgeProxy: boolean;
111
+ }
112
+
113
+ /** `GET /v1/remote-web/pairing-requests` success response body (200). */
114
+ export interface RemoteWebPairingRequestListResponse {
115
+ requests: RemoteWebPairingRequestSummary[];
116
+ }
117
+
118
+ /**
119
+ * Request body for the pairing-request approve and deny routes. The approve
120
+ * route's success body reuses {@link RemoteWebPairingVerificationResponse}.
121
+ */
122
+ export interface RemoteWebPairingRequestActionRequest {
123
+ requestId: string;
124
+ }
125
+
126
+ /** `POST /v1/remote-web/pairing-requests/deny` success response body (200). */
127
+ export interface RemoteWebPairingRequestDenyResponse {
128
+ status: "denied";
129
+ }
130
+
71
131
  /** `POST /v1/remote-web/pairing-token` request body. */
72
132
  export interface RemoteWebPairingTokenRequest {
73
133
  /** The `deviceCode` from the challenge. */
@@ -11,6 +11,12 @@
11
11
  * (`gateway/src/http/routes/remote-web-pairing-verification.ts`)
12
12
  * - `POST /v1/remote-web/pairing-token` poll + exchange device code
13
13
  * (`gateway/src/http/routes/remote-web-pairing-token.ts`)
14
+ * - `GET /v1/remote-web/pairing-requests` list pending challenges
15
+ * (loopback-only)
16
+ * - `POST /v1/remote-web/pairing-requests/approve` approve by request id
17
+ * (loopback-only)
18
+ * - `POST /v1/remote-web/pairing-requests/deny` deny (delete) by request id
19
+ * (loopback-only)
14
20
  *
15
21
  * These shapes mirror those handlers' request/response bodies exactly so the
16
22
  * gateway, the `vellum pair` CLI (`cli/src/commands/pair.ts`), and the web SPA
@@ -68,6 +74,60 @@ export interface RemoteWebPairingVerificationResponse {
68
74
  expiresAt: string;
69
75
  }
70
76
 
77
+ /**
78
+ * One pending challenge as shown on a host approval surface.
79
+ *
80
+ * The requesting device already sees the plaintext `userCode` in its own
81
+ * challenge response ({@link RemoteWebPairingChallengeResponse.userCode});
82
+ * the loopback-gated list route is the only host-side re-exposure. Displaying
83
+ * it there is what lets the approver match the code against the requesting
84
+ * device's screen: the device-flow anti-phishing binding.
85
+ */
86
+ export interface RemoteWebPairingRequestSummary {
87
+ /** Opaque server-side id used to approve or deny this request. */
88
+ requestId: string;
89
+ /** The human-readable code the requesting device is displaying (e.g. "ABCD-EFGH"). */
90
+ userCode: string;
91
+ /** Public base URL the challenge was minted for. */
92
+ publicBaseUrl: string;
93
+ /** ISO-8601 instant the challenge was minted. */
94
+ requestedAt: string;
95
+ /** ISO-8601 instant the challenge expires. */
96
+ expiresAt: string;
97
+ /**
98
+ * Client IP of the mint request: the loopback/host address when minted
99
+ * locally, or the edge-observed client address when the mint arrived
100
+ * through the nginx tunnel edge (which stamps it via `proxy_set_header`,
101
+ * so a remote client cannot smuggle a value).
102
+ */
103
+ requesterIp: string;
104
+ /** User-Agent header of the mint request, or null when absent. */
105
+ requesterUserAgent: string | null;
106
+ /**
107
+ * Whether the mint arrived through the public tunnel edge rather than the
108
+ * host itself.
109
+ */
110
+ viaEdgeProxy: boolean;
111
+ }
112
+
113
+ /** `GET /v1/remote-web/pairing-requests` success response body (200). */
114
+ export interface RemoteWebPairingRequestListResponse {
115
+ requests: RemoteWebPairingRequestSummary[];
116
+ }
117
+
118
+ /**
119
+ * Request body for the pairing-request approve and deny routes. The approve
120
+ * route's success body reuses {@link RemoteWebPairingVerificationResponse}.
121
+ */
122
+ export interface RemoteWebPairingRequestActionRequest {
123
+ requestId: string;
124
+ }
125
+
126
+ /** `POST /v1/remote-web/pairing-requests/deny` success response body (200). */
127
+ export interface RemoteWebPairingRequestDenyResponse {
128
+ status: "denied";
129
+ }
130
+
71
131
  /** `POST /v1/remote-web/pairing-token` request body. */
72
132
  export interface RemoteWebPairingTokenRequest {
73
133
  /** The `deviceCode` from the challenge. */
@@ -11,6 +11,12 @@
11
11
  * (`gateway/src/http/routes/remote-web-pairing-verification.ts`)
12
12
  * - `POST /v1/remote-web/pairing-token` poll + exchange device code
13
13
  * (`gateway/src/http/routes/remote-web-pairing-token.ts`)
14
+ * - `GET /v1/remote-web/pairing-requests` list pending challenges
15
+ * (loopback-only)
16
+ * - `POST /v1/remote-web/pairing-requests/approve` approve by request id
17
+ * (loopback-only)
18
+ * - `POST /v1/remote-web/pairing-requests/deny` deny (delete) by request id
19
+ * (loopback-only)
14
20
  *
15
21
  * These shapes mirror those handlers' request/response bodies exactly so the
16
22
  * gateway, the `vellum pair` CLI (`cli/src/commands/pair.ts`), and the web SPA
@@ -68,6 +74,60 @@ export interface RemoteWebPairingVerificationResponse {
68
74
  expiresAt: string;
69
75
  }
70
76
 
77
+ /**
78
+ * One pending challenge as shown on a host approval surface.
79
+ *
80
+ * The requesting device already sees the plaintext `userCode` in its own
81
+ * challenge response ({@link RemoteWebPairingChallengeResponse.userCode});
82
+ * the loopback-gated list route is the only host-side re-exposure. Displaying
83
+ * it there is what lets the approver match the code against the requesting
84
+ * device's screen: the device-flow anti-phishing binding.
85
+ */
86
+ export interface RemoteWebPairingRequestSummary {
87
+ /** Opaque server-side id used to approve or deny this request. */
88
+ requestId: string;
89
+ /** The human-readable code the requesting device is displaying (e.g. "ABCD-EFGH"). */
90
+ userCode: string;
91
+ /** Public base URL the challenge was minted for. */
92
+ publicBaseUrl: string;
93
+ /** ISO-8601 instant the challenge was minted. */
94
+ requestedAt: string;
95
+ /** ISO-8601 instant the challenge expires. */
96
+ expiresAt: string;
97
+ /**
98
+ * Client IP of the mint request: the loopback/host address when minted
99
+ * locally, or the edge-observed client address when the mint arrived
100
+ * through the nginx tunnel edge (which stamps it via `proxy_set_header`,
101
+ * so a remote client cannot smuggle a value).
102
+ */
103
+ requesterIp: string;
104
+ /** User-Agent header of the mint request, or null when absent. */
105
+ requesterUserAgent: string | null;
106
+ /**
107
+ * Whether the mint arrived through the public tunnel edge rather than the
108
+ * host itself.
109
+ */
110
+ viaEdgeProxy: boolean;
111
+ }
112
+
113
+ /** `GET /v1/remote-web/pairing-requests` success response body (200). */
114
+ export interface RemoteWebPairingRequestListResponse {
115
+ requests: RemoteWebPairingRequestSummary[];
116
+ }
117
+
118
+ /**
119
+ * Request body for the pairing-request approve and deny routes. The approve
120
+ * route's success body reuses {@link RemoteWebPairingVerificationResponse}.
121
+ */
122
+ export interface RemoteWebPairingRequestActionRequest {
123
+ requestId: string;
124
+ }
125
+
126
+ /** `POST /v1/remote-web/pairing-requests/deny` success response body (200). */
127
+ export interface RemoteWebPairingRequestDenyResponse {
128
+ status: "denied";
129
+ }
130
+
71
131
  /** `POST /v1/remote-web/pairing-token` request body. */
72
132
  export interface RemoteWebPairingTokenRequest {
73
133
  /** The `deviceCode` from the challenge. */
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@vellumai/assistant",
3
- "version": "0.11.4-staging.3",
3
+ "version": "0.11.4",
4
4
  "license": "MIT",
5
5
  "type": "module",
6
6
  "exports": {
@@ -713,3 +713,212 @@ describe("loadFromDb tool-result truncation", () => {
713
713
  ).toBe(preStubbed);
714
714
  });
715
715
  });
716
+
717
+ /**
718
+ * A guardian card is written into the conversation the request came from by
719
+ * the notification pairing layer, straight to the DB, while that
720
+ * conversation's turn is parked awaiting the approval. On the next load it
721
+ * lands between the parked turn's `tool_use` and its `tool_result`.
722
+ *
723
+ * The card is recognized by its own `ui_surface` id rather than a stored
724
+ * marker, so rows written before this rule existed are dropped on the same
725
+ * terms as new ones.
726
+ */
727
+ describe("guardian card rows are not replayed as conversation history", () => {
728
+ const baseConversation = {
729
+ id: "conv-1",
730
+ contextSummary: null,
731
+ contextCompactedMessageCount: 0,
732
+ totalInputTokens: 0,
733
+ totalOutputTokens: 0,
734
+ totalEstimatedCost: 0,
735
+ };
736
+
737
+ /** The parked turn: an approval-gated `tool_use` and the result it got. */
738
+ const toolUseRow = {
739
+ id: "m-use",
740
+ role: "assistant",
741
+ content: [
742
+ { type: "text", text: "Running that now." },
743
+ { type: "tool_use", id: "tu_1", name: "bash", input: { cmd: "ls" } },
744
+ ],
745
+ };
746
+ const toolResultRow = {
747
+ id: "m-result",
748
+ role: "user",
749
+ content: [
750
+ { type: "tool_result", tool_use_id: "tu_1", content: "REAL RESULT" },
751
+ ],
752
+ };
753
+
754
+ /** What `buildApprovalCardBlocks` persists: the surface, then its text. */
755
+ function cardRow(surfaceId: string) {
756
+ return {
757
+ id: "m-card",
758
+ role: "assistant",
759
+ content: [
760
+ {
761
+ type: "ui_surface",
762
+ surfaceId,
763
+ surfaceType: "card",
764
+ title: "Tool Approval",
765
+ data: {},
766
+ display: "inline",
767
+ },
768
+ { type: "text", text: "Tool Approval\nbash - requested by Alice" },
769
+ ],
770
+ };
771
+ }
772
+
773
+ function allBlocks(messages: Message[]) {
774
+ return messages.flatMap((m) => m.content);
775
+ }
776
+
777
+ function toolResultsFor(messages: Message[], toolUseId: string) {
778
+ return allBlocks(messages).filter(
779
+ (b) => b.type === "tool_result" && b.tool_use_id === toolUseId,
780
+ );
781
+ }
782
+
783
+ function serialized(value: unknown): string {
784
+ return JSON.stringify(value);
785
+ }
786
+
787
+ beforeEach(() => {
788
+ nextMockMessageId = 1;
789
+ mockConversation = { ...baseConversation };
790
+ });
791
+
792
+ // Sensitivity check for the drop below: an ordinary assistant turn in the
793
+ // same position still corrupts the pair. If the rule ever widened past
794
+ // guardian cards, this test fails first and says so.
795
+ test("an ordinary assistant turn between a tool_use and its result still corrupts it", async () => {
796
+ mockDbMessages = [
797
+ toolUseRow,
798
+ {
799
+ id: "m-card",
800
+ role: "assistant",
801
+ content: [{ type: "text", text: "Still working on it." }],
802
+ },
803
+ toolResultRow,
804
+ ];
805
+
806
+ const conversation = makeConversation();
807
+ await conversation.loadFromDb();
808
+ const messages = conversation.getMessages();
809
+
810
+ expect(serialized(messages)).toContain("tool result missing from history");
811
+ expect(serialized(messages)).toContain("[orphaned tool_result for tu_1]");
812
+ expect(serialized(toolResultsFor(messages, "tu_1"))).not.toContain(
813
+ "REAL RESULT",
814
+ );
815
+ });
816
+
817
+ test("a guardian card is dropped and the real tool_result survives", async () => {
818
+ mockDbMessages = [
819
+ toolUseRow,
820
+ cardRow("tool-approval-req-1"),
821
+ toolResultRow,
822
+ ];
823
+
824
+ const conversation = makeConversation();
825
+ await conversation.loadFromDb();
826
+ const messages = conversation.getMessages();
827
+
828
+ // The pair is adjacent and intact: no stub, no downgrade, and exactly one
829
+ // tool_result for tu_1 carrying the tool's own output.
830
+ expect(serialized(messages)).not.toContain(
831
+ "tool result missing from history",
832
+ );
833
+ expect(serialized(messages)).not.toContain("[orphaned tool_result");
834
+ const results = toolResultsFor(messages, "tu_1");
835
+ expect(results).toHaveLength(1);
836
+ expect(results[0].type === "tool_result" && results[0].content).toBe(
837
+ "REAL RESULT",
838
+ );
839
+ expect(serialized(messages)).not.toContain("Tool Approval");
840
+ expect(messages).toHaveLength(2);
841
+ });
842
+
843
+ test("an access-request card is dropped on the same terms", async () => {
844
+ mockDbMessages = [
845
+ toolUseRow,
846
+ cardRow("access-request-req-9"),
847
+ toolResultRow,
848
+ ];
849
+
850
+ const conversation = makeConversation();
851
+ await conversation.loadFromDb();
852
+ const messages = conversation.getMessages();
853
+
854
+ expect(serialized(messages)).not.toContain("[orphaned tool_result");
855
+ expect(messages).toHaveLength(2);
856
+ });
857
+
858
+ test("a card as the tail row leaves history ending on a user message", async () => {
859
+ // The prefill shape: the approval was delivered and the turn never
860
+ // resumed, so the card is the last row. Replayed, it makes the history
861
+ // end on an assistant turn, which extended-thinking models reject with
862
+ // "does not support assistant message prefill".
863
+ mockDbMessages = [toolUseRow, cardRow("tool-approval-req-1")];
864
+
865
+ const conversation = makeConversation();
866
+ await conversation.loadFromDb();
867
+ const messages = conversation.getMessages();
868
+
869
+ expect(messages[messages.length - 1].role).toBe("user");
870
+ // The tail is repair's fill for the still-unanswered tool_use, which is
871
+ // correct here: the tool genuinely never ran.
872
+ expect(toolResultsFor(messages, "tu_1")).toHaveLength(1);
873
+ });
874
+
875
+ test("a dropped card still restores its ui_surface for action routing", async () => {
876
+ // Surface lifecycle and model context are different questions. The card is
877
+ // absent from history, but its Approve/Reject buttons must keep routing
878
+ // after a restart, which `findConversationBySurfaceId` resolves through
879
+ // surfaceState rebuilt at load.
880
+ mockDbMessages = [toolUseRow, cardRow("tool-approval-req-1")];
881
+
882
+ const conversation = makeConversation();
883
+ await conversation.loadFromDb();
884
+
885
+ expect(serialized(conversation.getMessages())).not.toContain(
886
+ "tool-approval-req-1",
887
+ );
888
+ expect(
889
+ (
890
+ conversation as unknown as { surfaceState: Map<string, unknown> }
891
+ ).surfaceState.has("tool-approval-req-1"),
892
+ ).toBe(true);
893
+ });
894
+
895
+ test("an assistant turn carrying a non-approval surface stays in history", async () => {
896
+ // A wake card is unshifted onto a real turn's own content. Its surface id
897
+ // uses a different prefix, and dropping that row would discard speech.
898
+ mockDbMessages = [
899
+ { id: "m-ask", role: "user", content: [{ type: "text", text: "hi" }] },
900
+ {
901
+ id: "m-wake",
902
+ role: "assistant",
903
+ content: [
904
+ {
905
+ type: "ui_surface",
906
+ surfaceId: "wake-conv-1-1700000000",
907
+ surfaceType: "card",
908
+ title: "Conversation Woke",
909
+ data: {},
910
+ display: "inline",
911
+ },
912
+ { type: "text", text: "Picking this back up." },
913
+ ],
914
+ },
915
+ ];
916
+
917
+ const conversation = makeConversation();
918
+ await conversation.loadFromDb();
919
+
920
+ expect(serialized(conversation.getMessages())).toContain(
921
+ "Picking this back up.",
922
+ );
923
+ });
924
+ });
@@ -4024,6 +4024,59 @@ describe("Slack channel chronological rendering — multi-thread", () => {
4024
4024
  expect(allText).not.toContain("private guardian-only context");
4025
4025
  });
4026
4026
 
4027
+ // Slack builds provider history from rows rather than `this.messages`, so a
4028
+ // guardian card dropped only at `Conversation.loadFromDb` would still reach
4029
+ // the model here -- the channel the incident happened on.
4030
+ test("loadSlackChronologicalContext drops guardian card rows", () => {
4031
+ const caps: ChannelCapabilities = {
4032
+ channel: "slack",
4033
+ dashboardCapable: false,
4034
+ supportsDynamicUi: false,
4035
+ supportsVoiceInput: false,
4036
+ chatType: "channel",
4037
+ };
4038
+ const cardRow: MessageRow = {
4039
+ id: "m-card",
4040
+ conversationId: "conv-1",
4041
+ role: "assistant",
4042
+ content: [
4043
+ {
4044
+ type: "ui_surface",
4045
+ surfaceId: "tool-approval-req-1",
4046
+ surfaceType: "card",
4047
+ title: "Tool Approval",
4048
+ data: {},
4049
+ display: "inline",
4050
+ },
4051
+ { type: "text", text: "Tool Approval card body" },
4052
+ ],
4053
+ createdAt: 1700000035_000,
4054
+ metadata: null,
4055
+ clientMessageId: null,
4056
+ finalized: 1,
4057
+ } as MessageRow;
4058
+ const rows: MessageRow[] = [
4059
+ userRow({
4060
+ id: "asked",
4061
+ createdAt: 1700000030_000,
4062
+ text: "please run it",
4063
+ slackMeta: buildSlackMeta({ channelTs: T2, displayName: "carol" }),
4064
+ }),
4065
+ cardRow,
4066
+ ];
4067
+
4068
+ const result = loadSlackChronologicalContext("conv-1", caps, {
4069
+ loader: () => rows,
4070
+ trustClass: "guardian",
4071
+ });
4072
+
4073
+ expect(result).not.toBeNull();
4074
+ const renderedText = JSON.stringify(result!.messages);
4075
+ expect(renderedText).toContain("please run it");
4076
+ expect(renderedText).not.toContain("Tool Approval card body");
4077
+ expect(renderedText).not.toContain("tool-approval-req-1");
4078
+ });
4079
+
4027
4080
  test("loadSlackChronologicalContext preserves summary and filters by Slack watermark", () => {
4028
4081
  const caps: ChannelCapabilities = {
4029
4082
  channel: "slack",