@vellumai/assistant 0.11.4-staging.3 → 0.11.4-staging.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/docs/guardian-request-flow.md +35 -0
- package/node_modules/@vellumai/ces-client/node_modules/@vellumai/service-contracts/src/remote-web-pairing.ts +60 -0
- package/node_modules/@vellumai/gateway-client/node_modules/@vellumai/service-contracts/src/remote-web-pairing.ts +60 -0
- package/node_modules/@vellumai/service-contracts/src/remote-web-pairing.ts +60 -0
- package/package.json +1 -1
- package/src/__tests__/conversation-load-history-repair.test.ts +209 -0
- package/src/__tests__/conversation-runtime-assembly.test.ts +53 -0
- package/src/__tests__/file-ops-service.test.ts +163 -30
- package/src/__tests__/filesystem-tools.test.ts +23 -24
- package/src/__tests__/host-file-read-tool.test.ts +16 -19
- package/src/__tests__/tool-executor.test.ts +5 -1
- package/src/api/events/host-file.ts +2 -2
- package/src/daemon/conversation-runtime-assembly.ts +9 -2
- package/src/daemon/conversation.ts +23 -7
- package/src/notifications/AGENTS.md +2 -0
- package/src/notifications/approval-card-data.ts +33 -0
- package/src/subagent/types.ts +7 -7
- package/src/tools/__tests__/tool-input-schemas.test.ts +7 -7
- package/src/tools/filesystem/read.ts +27 -10
- package/src/tools/host-filesystem/read.ts +27 -15
- package/src/tools/shared/filesystem/file-ops-service.ts +63 -35
- package/src/tools/shared/filesystem/legacy-read-args.ts +22 -0
- package/src/tools/shared/filesystem/types.ts +5 -5
|
@@ -54,6 +54,41 @@ anti-pattern was retired in #35642 and again in the ask_question redesign).
|
|
|
54
54
|
| Kind-specific follow-through | resolver registry (`kind` → resolver) | switch statements in the router |
|
|
55
55
|
| Reply understanding | guardian reply router (codes, buttons, reactions, modes) | per-feature inbound intercepts |
|
|
56
56
|
|
|
57
|
+
## Cards are not conversation history
|
|
58
|
+
|
|
59
|
+
`pairDeliveryWithConversation` persists one message row per delivery so the
|
|
60
|
+
card renders and deep-links. For a guardian card that row is addressed to a
|
|
61
|
+
conversation the request is _about_, not one the assistant is speaking in:
|
|
62
|
+
`buildVellumCardAffinity` pins the vellum card to the originating
|
|
63
|
+
conversation, and a channel card lands in whatever conversation the guardian's
|
|
64
|
+
chat binds to. Either way the row is written straight to the DB by the
|
|
65
|
+
notification pipeline, so the live turn's in-memory history never sees it.
|
|
66
|
+
|
|
67
|
+
That row must never be replayed to the model. The conversation it lands in is
|
|
68
|
+
typically parked mid-approval, with its last assistant message carrying the
|
|
69
|
+
`tool_use` still waiting on this very decision. Replayed, the card sits
|
|
70
|
+
between that `tool_use` and its `tool_result`; history repair reads the pair
|
|
71
|
+
as broken, synthesizes a stub result, and downgrades the real one to text. A
|
|
72
|
+
card left as the tail row instead ends the history on an assistant message,
|
|
73
|
+
which extended-thinking models reject outright ("does not support assistant
|
|
74
|
+
message prefill").
|
|
75
|
+
|
|
76
|
+
`isGuardianCardRow` (`notifications/approval-card-data.ts`) is the one
|
|
77
|
+
definition of which rows those are, read off the card's own `ui_surface` id
|
|
78
|
+
using the same prefixes `approvalCardSurfaceId` recomputes for withdrawal. It
|
|
79
|
+
is derived rather than stored so a row written before the rule existed is
|
|
80
|
+
recognized on the same terms as a new one, with no marker to backfill.
|
|
81
|
+
|
|
82
|
+
**Both history assemblers must consult it.** `Conversation.loadFromDb` builds
|
|
83
|
+
`this.messages`, but Slack conversations do not use that list:
|
|
84
|
+
`loadSlackChronologicalContext` re-reads the rows. A rule applied to only one
|
|
85
|
+
exempts the other channel. Each applies it _after_ its own compaction boundary,
|
|
86
|
+
since both boundaries are computed against unfiltered row lists.
|
|
87
|
+
|
|
88
|
+
Surface state is deliberately NOT filtered this way. `restoreSurfaceStateFromHistory`
|
|
89
|
+
takes the pre-filter window, because a card's Approve/Reject buttons must keep
|
|
90
|
+
routing after a restart even though the card is absent from the model's history.
|
|
91
|
+
|
|
57
92
|
Two instruction modes exist per request kind (`notifications/guardian-question-mode.ts`):
|
|
58
93
|
**approval** ("CODE approve" / approve–reject buttons) and **answer**
|
|
59
94
|
("CODE <your answer>" / option buttons). `pending_question` is answer-mode.
|
|
@@ -11,6 +11,12 @@
|
|
|
11
11
|
* (`gateway/src/http/routes/remote-web-pairing-verification.ts`)
|
|
12
12
|
* - `POST /v1/remote-web/pairing-token` poll + exchange device code
|
|
13
13
|
* (`gateway/src/http/routes/remote-web-pairing-token.ts`)
|
|
14
|
+
* - `GET /v1/remote-web/pairing-requests` list pending challenges
|
|
15
|
+
* (loopback-only)
|
|
16
|
+
* - `POST /v1/remote-web/pairing-requests/approve` approve by request id
|
|
17
|
+
* (loopback-only)
|
|
18
|
+
* - `POST /v1/remote-web/pairing-requests/deny` deny (delete) by request id
|
|
19
|
+
* (loopback-only)
|
|
14
20
|
*
|
|
15
21
|
* These shapes mirror those handlers' request/response bodies exactly so the
|
|
16
22
|
* gateway, the `vellum pair` CLI (`cli/src/commands/pair.ts`), and the web SPA
|
|
@@ -68,6 +74,60 @@ export interface RemoteWebPairingVerificationResponse {
|
|
|
68
74
|
expiresAt: string;
|
|
69
75
|
}
|
|
70
76
|
|
|
77
|
+
/**
|
|
78
|
+
* One pending challenge as shown on a host approval surface.
|
|
79
|
+
*
|
|
80
|
+
* The requesting device already sees the plaintext `userCode` in its own
|
|
81
|
+
* challenge response ({@link RemoteWebPairingChallengeResponse.userCode});
|
|
82
|
+
* the loopback-gated list route is the only host-side re-exposure. Displaying
|
|
83
|
+
* it there is what lets the approver match the code against the requesting
|
|
84
|
+
* device's screen: the device-flow anti-phishing binding.
|
|
85
|
+
*/
|
|
86
|
+
export interface RemoteWebPairingRequestSummary {
|
|
87
|
+
/** Opaque server-side id used to approve or deny this request. */
|
|
88
|
+
requestId: string;
|
|
89
|
+
/** The human-readable code the requesting device is displaying (e.g. "ABCD-EFGH"). */
|
|
90
|
+
userCode: string;
|
|
91
|
+
/** Public base URL the challenge was minted for. */
|
|
92
|
+
publicBaseUrl: string;
|
|
93
|
+
/** ISO-8601 instant the challenge was minted. */
|
|
94
|
+
requestedAt: string;
|
|
95
|
+
/** ISO-8601 instant the challenge expires. */
|
|
96
|
+
expiresAt: string;
|
|
97
|
+
/**
|
|
98
|
+
* Client IP of the mint request: the loopback/host address when minted
|
|
99
|
+
* locally, or the edge-observed client address when the mint arrived
|
|
100
|
+
* through the nginx tunnel edge (which stamps it via `proxy_set_header`,
|
|
101
|
+
* so a remote client cannot smuggle a value).
|
|
102
|
+
*/
|
|
103
|
+
requesterIp: string;
|
|
104
|
+
/** User-Agent header of the mint request, or null when absent. */
|
|
105
|
+
requesterUserAgent: string | null;
|
|
106
|
+
/**
|
|
107
|
+
* Whether the mint arrived through the public tunnel edge rather than the
|
|
108
|
+
* host itself.
|
|
109
|
+
*/
|
|
110
|
+
viaEdgeProxy: boolean;
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
/** `GET /v1/remote-web/pairing-requests` success response body (200). */
|
|
114
|
+
export interface RemoteWebPairingRequestListResponse {
|
|
115
|
+
requests: RemoteWebPairingRequestSummary[];
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
/**
|
|
119
|
+
* Request body for the pairing-request approve and deny routes. The approve
|
|
120
|
+
* route's success body reuses {@link RemoteWebPairingVerificationResponse}.
|
|
121
|
+
*/
|
|
122
|
+
export interface RemoteWebPairingRequestActionRequest {
|
|
123
|
+
requestId: string;
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
/** `POST /v1/remote-web/pairing-requests/deny` success response body (200). */
|
|
127
|
+
export interface RemoteWebPairingRequestDenyResponse {
|
|
128
|
+
status: "denied";
|
|
129
|
+
}
|
|
130
|
+
|
|
71
131
|
/** `POST /v1/remote-web/pairing-token` request body. */
|
|
72
132
|
export interface RemoteWebPairingTokenRequest {
|
|
73
133
|
/** The `deviceCode` from the challenge. */
|
|
@@ -11,6 +11,12 @@
|
|
|
11
11
|
* (`gateway/src/http/routes/remote-web-pairing-verification.ts`)
|
|
12
12
|
* - `POST /v1/remote-web/pairing-token` poll + exchange device code
|
|
13
13
|
* (`gateway/src/http/routes/remote-web-pairing-token.ts`)
|
|
14
|
+
* - `GET /v1/remote-web/pairing-requests` list pending challenges
|
|
15
|
+
* (loopback-only)
|
|
16
|
+
* - `POST /v1/remote-web/pairing-requests/approve` approve by request id
|
|
17
|
+
* (loopback-only)
|
|
18
|
+
* - `POST /v1/remote-web/pairing-requests/deny` deny (delete) by request id
|
|
19
|
+
* (loopback-only)
|
|
14
20
|
*
|
|
15
21
|
* These shapes mirror those handlers' request/response bodies exactly so the
|
|
16
22
|
* gateway, the `vellum pair` CLI (`cli/src/commands/pair.ts`), and the web SPA
|
|
@@ -68,6 +74,60 @@ export interface RemoteWebPairingVerificationResponse {
|
|
|
68
74
|
expiresAt: string;
|
|
69
75
|
}
|
|
70
76
|
|
|
77
|
+
/**
|
|
78
|
+
* One pending challenge as shown on a host approval surface.
|
|
79
|
+
*
|
|
80
|
+
* The requesting device already sees the plaintext `userCode` in its own
|
|
81
|
+
* challenge response ({@link RemoteWebPairingChallengeResponse.userCode});
|
|
82
|
+
* the loopback-gated list route is the only host-side re-exposure. Displaying
|
|
83
|
+
* it there is what lets the approver match the code against the requesting
|
|
84
|
+
* device's screen: the device-flow anti-phishing binding.
|
|
85
|
+
*/
|
|
86
|
+
export interface RemoteWebPairingRequestSummary {
|
|
87
|
+
/** Opaque server-side id used to approve or deny this request. */
|
|
88
|
+
requestId: string;
|
|
89
|
+
/** The human-readable code the requesting device is displaying (e.g. "ABCD-EFGH"). */
|
|
90
|
+
userCode: string;
|
|
91
|
+
/** Public base URL the challenge was minted for. */
|
|
92
|
+
publicBaseUrl: string;
|
|
93
|
+
/** ISO-8601 instant the challenge was minted. */
|
|
94
|
+
requestedAt: string;
|
|
95
|
+
/** ISO-8601 instant the challenge expires. */
|
|
96
|
+
expiresAt: string;
|
|
97
|
+
/**
|
|
98
|
+
* Client IP of the mint request: the loopback/host address when minted
|
|
99
|
+
* locally, or the edge-observed client address when the mint arrived
|
|
100
|
+
* through the nginx tunnel edge (which stamps it via `proxy_set_header`,
|
|
101
|
+
* so a remote client cannot smuggle a value).
|
|
102
|
+
*/
|
|
103
|
+
requesterIp: string;
|
|
104
|
+
/** User-Agent header of the mint request, or null when absent. */
|
|
105
|
+
requesterUserAgent: string | null;
|
|
106
|
+
/**
|
|
107
|
+
* Whether the mint arrived through the public tunnel edge rather than the
|
|
108
|
+
* host itself.
|
|
109
|
+
*/
|
|
110
|
+
viaEdgeProxy: boolean;
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
/** `GET /v1/remote-web/pairing-requests` success response body (200). */
|
|
114
|
+
export interface RemoteWebPairingRequestListResponse {
|
|
115
|
+
requests: RemoteWebPairingRequestSummary[];
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
/**
|
|
119
|
+
* Request body for the pairing-request approve and deny routes. The approve
|
|
120
|
+
* route's success body reuses {@link RemoteWebPairingVerificationResponse}.
|
|
121
|
+
*/
|
|
122
|
+
export interface RemoteWebPairingRequestActionRequest {
|
|
123
|
+
requestId: string;
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
/** `POST /v1/remote-web/pairing-requests/deny` success response body (200). */
|
|
127
|
+
export interface RemoteWebPairingRequestDenyResponse {
|
|
128
|
+
status: "denied";
|
|
129
|
+
}
|
|
130
|
+
|
|
71
131
|
/** `POST /v1/remote-web/pairing-token` request body. */
|
|
72
132
|
export interface RemoteWebPairingTokenRequest {
|
|
73
133
|
/** The `deviceCode` from the challenge. */
|
|
@@ -11,6 +11,12 @@
|
|
|
11
11
|
* (`gateway/src/http/routes/remote-web-pairing-verification.ts`)
|
|
12
12
|
* - `POST /v1/remote-web/pairing-token` poll + exchange device code
|
|
13
13
|
* (`gateway/src/http/routes/remote-web-pairing-token.ts`)
|
|
14
|
+
* - `GET /v1/remote-web/pairing-requests` list pending challenges
|
|
15
|
+
* (loopback-only)
|
|
16
|
+
* - `POST /v1/remote-web/pairing-requests/approve` approve by request id
|
|
17
|
+
* (loopback-only)
|
|
18
|
+
* - `POST /v1/remote-web/pairing-requests/deny` deny (delete) by request id
|
|
19
|
+
* (loopback-only)
|
|
14
20
|
*
|
|
15
21
|
* These shapes mirror those handlers' request/response bodies exactly so the
|
|
16
22
|
* gateway, the `vellum pair` CLI (`cli/src/commands/pair.ts`), and the web SPA
|
|
@@ -68,6 +74,60 @@ export interface RemoteWebPairingVerificationResponse {
|
|
|
68
74
|
expiresAt: string;
|
|
69
75
|
}
|
|
70
76
|
|
|
77
|
+
/**
|
|
78
|
+
* One pending challenge as shown on a host approval surface.
|
|
79
|
+
*
|
|
80
|
+
* The requesting device already sees the plaintext `userCode` in its own
|
|
81
|
+
* challenge response ({@link RemoteWebPairingChallengeResponse.userCode});
|
|
82
|
+
* the loopback-gated list route is the only host-side re-exposure. Displaying
|
|
83
|
+
* it there is what lets the approver match the code against the requesting
|
|
84
|
+
* device's screen: the device-flow anti-phishing binding.
|
|
85
|
+
*/
|
|
86
|
+
export interface RemoteWebPairingRequestSummary {
|
|
87
|
+
/** Opaque server-side id used to approve or deny this request. */
|
|
88
|
+
requestId: string;
|
|
89
|
+
/** The human-readable code the requesting device is displaying (e.g. "ABCD-EFGH"). */
|
|
90
|
+
userCode: string;
|
|
91
|
+
/** Public base URL the challenge was minted for. */
|
|
92
|
+
publicBaseUrl: string;
|
|
93
|
+
/** ISO-8601 instant the challenge was minted. */
|
|
94
|
+
requestedAt: string;
|
|
95
|
+
/** ISO-8601 instant the challenge expires. */
|
|
96
|
+
expiresAt: string;
|
|
97
|
+
/**
|
|
98
|
+
* Client IP of the mint request: the loopback/host address when minted
|
|
99
|
+
* locally, or the edge-observed client address when the mint arrived
|
|
100
|
+
* through the nginx tunnel edge (which stamps it via `proxy_set_header`,
|
|
101
|
+
* so a remote client cannot smuggle a value).
|
|
102
|
+
*/
|
|
103
|
+
requesterIp: string;
|
|
104
|
+
/** User-Agent header of the mint request, or null when absent. */
|
|
105
|
+
requesterUserAgent: string | null;
|
|
106
|
+
/**
|
|
107
|
+
* Whether the mint arrived through the public tunnel edge rather than the
|
|
108
|
+
* host itself.
|
|
109
|
+
*/
|
|
110
|
+
viaEdgeProxy: boolean;
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
/** `GET /v1/remote-web/pairing-requests` success response body (200). */
|
|
114
|
+
export interface RemoteWebPairingRequestListResponse {
|
|
115
|
+
requests: RemoteWebPairingRequestSummary[];
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
/**
|
|
119
|
+
* Request body for the pairing-request approve and deny routes. The approve
|
|
120
|
+
* route's success body reuses {@link RemoteWebPairingVerificationResponse}.
|
|
121
|
+
*/
|
|
122
|
+
export interface RemoteWebPairingRequestActionRequest {
|
|
123
|
+
requestId: string;
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
/** `POST /v1/remote-web/pairing-requests/deny` success response body (200). */
|
|
127
|
+
export interface RemoteWebPairingRequestDenyResponse {
|
|
128
|
+
status: "denied";
|
|
129
|
+
}
|
|
130
|
+
|
|
71
131
|
/** `POST /v1/remote-web/pairing-token` request body. */
|
|
72
132
|
export interface RemoteWebPairingTokenRequest {
|
|
73
133
|
/** The `deviceCode` from the challenge. */
|
package/package.json
CHANGED
|
@@ -713,3 +713,212 @@ describe("loadFromDb tool-result truncation", () => {
|
|
|
713
713
|
).toBe(preStubbed);
|
|
714
714
|
});
|
|
715
715
|
});
|
|
716
|
+
|
|
717
|
+
/**
|
|
718
|
+
* A guardian card is written into the conversation the request came from by
|
|
719
|
+
* the notification pairing layer, straight to the DB, while that
|
|
720
|
+
* conversation's turn is parked awaiting the approval. On the next load it
|
|
721
|
+
* lands between the parked turn's `tool_use` and its `tool_result`.
|
|
722
|
+
*
|
|
723
|
+
* The card is recognized by its own `ui_surface` id rather than a stored
|
|
724
|
+
* marker, so rows written before this rule existed are dropped on the same
|
|
725
|
+
* terms as new ones.
|
|
726
|
+
*/
|
|
727
|
+
describe("guardian card rows are not replayed as conversation history", () => {
|
|
728
|
+
const baseConversation = {
|
|
729
|
+
id: "conv-1",
|
|
730
|
+
contextSummary: null,
|
|
731
|
+
contextCompactedMessageCount: 0,
|
|
732
|
+
totalInputTokens: 0,
|
|
733
|
+
totalOutputTokens: 0,
|
|
734
|
+
totalEstimatedCost: 0,
|
|
735
|
+
};
|
|
736
|
+
|
|
737
|
+
/** The parked turn: an approval-gated `tool_use` and the result it got. */
|
|
738
|
+
const toolUseRow = {
|
|
739
|
+
id: "m-use",
|
|
740
|
+
role: "assistant",
|
|
741
|
+
content: [
|
|
742
|
+
{ type: "text", text: "Running that now." },
|
|
743
|
+
{ type: "tool_use", id: "tu_1", name: "bash", input: { cmd: "ls" } },
|
|
744
|
+
],
|
|
745
|
+
};
|
|
746
|
+
const toolResultRow = {
|
|
747
|
+
id: "m-result",
|
|
748
|
+
role: "user",
|
|
749
|
+
content: [
|
|
750
|
+
{ type: "tool_result", tool_use_id: "tu_1", content: "REAL RESULT" },
|
|
751
|
+
],
|
|
752
|
+
};
|
|
753
|
+
|
|
754
|
+
/** What `buildApprovalCardBlocks` persists: the surface, then its text. */
|
|
755
|
+
function cardRow(surfaceId: string) {
|
|
756
|
+
return {
|
|
757
|
+
id: "m-card",
|
|
758
|
+
role: "assistant",
|
|
759
|
+
content: [
|
|
760
|
+
{
|
|
761
|
+
type: "ui_surface",
|
|
762
|
+
surfaceId,
|
|
763
|
+
surfaceType: "card",
|
|
764
|
+
title: "Tool Approval",
|
|
765
|
+
data: {},
|
|
766
|
+
display: "inline",
|
|
767
|
+
},
|
|
768
|
+
{ type: "text", text: "Tool Approval\nbash - requested by Alice" },
|
|
769
|
+
],
|
|
770
|
+
};
|
|
771
|
+
}
|
|
772
|
+
|
|
773
|
+
function allBlocks(messages: Message[]) {
|
|
774
|
+
return messages.flatMap((m) => m.content);
|
|
775
|
+
}
|
|
776
|
+
|
|
777
|
+
function toolResultsFor(messages: Message[], toolUseId: string) {
|
|
778
|
+
return allBlocks(messages).filter(
|
|
779
|
+
(b) => b.type === "tool_result" && b.tool_use_id === toolUseId,
|
|
780
|
+
);
|
|
781
|
+
}
|
|
782
|
+
|
|
783
|
+
function serialized(value: unknown): string {
|
|
784
|
+
return JSON.stringify(value);
|
|
785
|
+
}
|
|
786
|
+
|
|
787
|
+
beforeEach(() => {
|
|
788
|
+
nextMockMessageId = 1;
|
|
789
|
+
mockConversation = { ...baseConversation };
|
|
790
|
+
});
|
|
791
|
+
|
|
792
|
+
// Sensitivity check for the drop below: an ordinary assistant turn in the
|
|
793
|
+
// same position still corrupts the pair. If the rule ever widened past
|
|
794
|
+
// guardian cards, this test fails first and says so.
|
|
795
|
+
test("an ordinary assistant turn between a tool_use and its result still corrupts it", async () => {
|
|
796
|
+
mockDbMessages = [
|
|
797
|
+
toolUseRow,
|
|
798
|
+
{
|
|
799
|
+
id: "m-card",
|
|
800
|
+
role: "assistant",
|
|
801
|
+
content: [{ type: "text", text: "Still working on it." }],
|
|
802
|
+
},
|
|
803
|
+
toolResultRow,
|
|
804
|
+
];
|
|
805
|
+
|
|
806
|
+
const conversation = makeConversation();
|
|
807
|
+
await conversation.loadFromDb();
|
|
808
|
+
const messages = conversation.getMessages();
|
|
809
|
+
|
|
810
|
+
expect(serialized(messages)).toContain("tool result missing from history");
|
|
811
|
+
expect(serialized(messages)).toContain("[orphaned tool_result for tu_1]");
|
|
812
|
+
expect(serialized(toolResultsFor(messages, "tu_1"))).not.toContain(
|
|
813
|
+
"REAL RESULT",
|
|
814
|
+
);
|
|
815
|
+
});
|
|
816
|
+
|
|
817
|
+
test("a guardian card is dropped and the real tool_result survives", async () => {
|
|
818
|
+
mockDbMessages = [
|
|
819
|
+
toolUseRow,
|
|
820
|
+
cardRow("tool-approval-req-1"),
|
|
821
|
+
toolResultRow,
|
|
822
|
+
];
|
|
823
|
+
|
|
824
|
+
const conversation = makeConversation();
|
|
825
|
+
await conversation.loadFromDb();
|
|
826
|
+
const messages = conversation.getMessages();
|
|
827
|
+
|
|
828
|
+
// The pair is adjacent and intact: no stub, no downgrade, and exactly one
|
|
829
|
+
// tool_result for tu_1 carrying the tool's own output.
|
|
830
|
+
expect(serialized(messages)).not.toContain(
|
|
831
|
+
"tool result missing from history",
|
|
832
|
+
);
|
|
833
|
+
expect(serialized(messages)).not.toContain("[orphaned tool_result");
|
|
834
|
+
const results = toolResultsFor(messages, "tu_1");
|
|
835
|
+
expect(results).toHaveLength(1);
|
|
836
|
+
expect(results[0].type === "tool_result" && results[0].content).toBe(
|
|
837
|
+
"REAL RESULT",
|
|
838
|
+
);
|
|
839
|
+
expect(serialized(messages)).not.toContain("Tool Approval");
|
|
840
|
+
expect(messages).toHaveLength(2);
|
|
841
|
+
});
|
|
842
|
+
|
|
843
|
+
test("an access-request card is dropped on the same terms", async () => {
|
|
844
|
+
mockDbMessages = [
|
|
845
|
+
toolUseRow,
|
|
846
|
+
cardRow("access-request-req-9"),
|
|
847
|
+
toolResultRow,
|
|
848
|
+
];
|
|
849
|
+
|
|
850
|
+
const conversation = makeConversation();
|
|
851
|
+
await conversation.loadFromDb();
|
|
852
|
+
const messages = conversation.getMessages();
|
|
853
|
+
|
|
854
|
+
expect(serialized(messages)).not.toContain("[orphaned tool_result");
|
|
855
|
+
expect(messages).toHaveLength(2);
|
|
856
|
+
});
|
|
857
|
+
|
|
858
|
+
test("a card as the tail row leaves history ending on a user message", async () => {
|
|
859
|
+
// The prefill shape: the approval was delivered and the turn never
|
|
860
|
+
// resumed, so the card is the last row. Replayed, it makes the history
|
|
861
|
+
// end on an assistant turn, which extended-thinking models reject with
|
|
862
|
+
// "does not support assistant message prefill".
|
|
863
|
+
mockDbMessages = [toolUseRow, cardRow("tool-approval-req-1")];
|
|
864
|
+
|
|
865
|
+
const conversation = makeConversation();
|
|
866
|
+
await conversation.loadFromDb();
|
|
867
|
+
const messages = conversation.getMessages();
|
|
868
|
+
|
|
869
|
+
expect(messages[messages.length - 1].role).toBe("user");
|
|
870
|
+
// The tail is repair's fill for the still-unanswered tool_use, which is
|
|
871
|
+
// correct here: the tool genuinely never ran.
|
|
872
|
+
expect(toolResultsFor(messages, "tu_1")).toHaveLength(1);
|
|
873
|
+
});
|
|
874
|
+
|
|
875
|
+
test("a dropped card still restores its ui_surface for action routing", async () => {
|
|
876
|
+
// Surface lifecycle and model context are different questions. The card is
|
|
877
|
+
// absent from history, but its Approve/Reject buttons must keep routing
|
|
878
|
+
// after a restart, which `findConversationBySurfaceId` resolves through
|
|
879
|
+
// surfaceState rebuilt at load.
|
|
880
|
+
mockDbMessages = [toolUseRow, cardRow("tool-approval-req-1")];
|
|
881
|
+
|
|
882
|
+
const conversation = makeConversation();
|
|
883
|
+
await conversation.loadFromDb();
|
|
884
|
+
|
|
885
|
+
expect(serialized(conversation.getMessages())).not.toContain(
|
|
886
|
+
"tool-approval-req-1",
|
|
887
|
+
);
|
|
888
|
+
expect(
|
|
889
|
+
(
|
|
890
|
+
conversation as unknown as { surfaceState: Map<string, unknown> }
|
|
891
|
+
).surfaceState.has("tool-approval-req-1"),
|
|
892
|
+
).toBe(true);
|
|
893
|
+
});
|
|
894
|
+
|
|
895
|
+
test("an assistant turn carrying a non-approval surface stays in history", async () => {
|
|
896
|
+
// A wake card is unshifted onto a real turn's own content. Its surface id
|
|
897
|
+
// uses a different prefix, and dropping that row would discard speech.
|
|
898
|
+
mockDbMessages = [
|
|
899
|
+
{ id: "m-ask", role: "user", content: [{ type: "text", text: "hi" }] },
|
|
900
|
+
{
|
|
901
|
+
id: "m-wake",
|
|
902
|
+
role: "assistant",
|
|
903
|
+
content: [
|
|
904
|
+
{
|
|
905
|
+
type: "ui_surface",
|
|
906
|
+
surfaceId: "wake-conv-1-1700000000",
|
|
907
|
+
surfaceType: "card",
|
|
908
|
+
title: "Conversation Woke",
|
|
909
|
+
data: {},
|
|
910
|
+
display: "inline",
|
|
911
|
+
},
|
|
912
|
+
{ type: "text", text: "Picking this back up." },
|
|
913
|
+
],
|
|
914
|
+
},
|
|
915
|
+
];
|
|
916
|
+
|
|
917
|
+
const conversation = makeConversation();
|
|
918
|
+
await conversation.loadFromDb();
|
|
919
|
+
|
|
920
|
+
expect(serialized(conversation.getMessages())).toContain(
|
|
921
|
+
"Picking this back up.",
|
|
922
|
+
);
|
|
923
|
+
});
|
|
924
|
+
});
|
|
@@ -4024,6 +4024,59 @@ describe("Slack channel chronological rendering — multi-thread", () => {
|
|
|
4024
4024
|
expect(allText).not.toContain("private guardian-only context");
|
|
4025
4025
|
});
|
|
4026
4026
|
|
|
4027
|
+
// Slack builds provider history from rows rather than `this.messages`, so a
|
|
4028
|
+
// guardian card dropped only at `Conversation.loadFromDb` would still reach
|
|
4029
|
+
// the model here -- the channel the incident happened on.
|
|
4030
|
+
test("loadSlackChronologicalContext drops guardian card rows", () => {
|
|
4031
|
+
const caps: ChannelCapabilities = {
|
|
4032
|
+
channel: "slack",
|
|
4033
|
+
dashboardCapable: false,
|
|
4034
|
+
supportsDynamicUi: false,
|
|
4035
|
+
supportsVoiceInput: false,
|
|
4036
|
+
chatType: "channel",
|
|
4037
|
+
};
|
|
4038
|
+
const cardRow: MessageRow = {
|
|
4039
|
+
id: "m-card",
|
|
4040
|
+
conversationId: "conv-1",
|
|
4041
|
+
role: "assistant",
|
|
4042
|
+
content: [
|
|
4043
|
+
{
|
|
4044
|
+
type: "ui_surface",
|
|
4045
|
+
surfaceId: "tool-approval-req-1",
|
|
4046
|
+
surfaceType: "card",
|
|
4047
|
+
title: "Tool Approval",
|
|
4048
|
+
data: {},
|
|
4049
|
+
display: "inline",
|
|
4050
|
+
},
|
|
4051
|
+
{ type: "text", text: "Tool Approval card body" },
|
|
4052
|
+
],
|
|
4053
|
+
createdAt: 1700000035_000,
|
|
4054
|
+
metadata: null,
|
|
4055
|
+
clientMessageId: null,
|
|
4056
|
+
finalized: 1,
|
|
4057
|
+
} as MessageRow;
|
|
4058
|
+
const rows: MessageRow[] = [
|
|
4059
|
+
userRow({
|
|
4060
|
+
id: "asked",
|
|
4061
|
+
createdAt: 1700000030_000,
|
|
4062
|
+
text: "please run it",
|
|
4063
|
+
slackMeta: buildSlackMeta({ channelTs: T2, displayName: "carol" }),
|
|
4064
|
+
}),
|
|
4065
|
+
cardRow,
|
|
4066
|
+
];
|
|
4067
|
+
|
|
4068
|
+
const result = loadSlackChronologicalContext("conv-1", caps, {
|
|
4069
|
+
loader: () => rows,
|
|
4070
|
+
trustClass: "guardian",
|
|
4071
|
+
});
|
|
4072
|
+
|
|
4073
|
+
expect(result).not.toBeNull();
|
|
4074
|
+
const renderedText = JSON.stringify(result!.messages);
|
|
4075
|
+
expect(renderedText).toContain("please run it");
|
|
4076
|
+
expect(renderedText).not.toContain("Tool Approval card body");
|
|
4077
|
+
expect(renderedText).not.toContain("tool-approval-req-1");
|
|
4078
|
+
});
|
|
4079
|
+
|
|
4027
4080
|
test("loadSlackChronologicalContext preserves summary and filters by Slack watermark", () => {
|
|
4028
4081
|
const caps: ChannelCapabilities = {
|
|
4029
4082
|
channel: "slack",
|