@stigmer/server 3.38.2 → 3.38.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/domain/agentexecution/controller.js +7 -3
- package/dist/domain/agentexecution/controller.js.map +1 -1
- package/dist/domain/agentexecution/update-status.d.ts +9 -2
- package/dist/domain/agentexecution/update-status.d.ts.map +1 -1
- package/dist/domain/agentexecution/update-status.js +28 -5
- package/dist/domain/agentexecution/update-status.js.map +1 -1
- package/dist/domain/agentexecution/validate-thinking-mode.d.ts +31 -21
- package/dist/domain/agentexecution/validate-thinking-mode.d.ts.map +1 -1
- package/dist/domain/agentexecution/validate-thinking-mode.js +61 -18
- package/dist/domain/agentexecution/validate-thinking-mode.js.map +1 -1
- package/dist/domain/artifact/controller.d.ts.map +1 -1
- package/dist/domain/artifact/controller.js +9 -6
- package/dist/domain/artifact/controller.js.map +1 -1
- package/dist/domain/environment/controller.d.ts +2 -2
- package/dist/domain/environment/controller.js +7 -3
- package/dist/domain/environment/controller.js.map +1 -1
- package/dist/domain/environment/steps.d.ts +2 -1
- package/dist/domain/environment/steps.d.ts.map +1 -1
- package/dist/domain/environment/steps.js +10 -7
- package/dist/domain/environment/steps.js.map +1 -1
- package/dist/domain/workflow/registry/data/model-registry.json +23 -14
- package/dist/domain/workflow/registry/model-registry-store.d.ts +14 -3
- package/dist/domain/workflow/registry/model-registry-store.d.ts.map +1 -1
- package/dist/domain/workflow/registry/model-registry-store.js +14 -3
- package/dist/domain/workflow/registry/model-registry-store.js.map +1 -1
- package/package.json +6 -6
- package/src/domain/agentexecution/__tests__/run-gate.test.ts +42 -1
- package/src/domain/agentexecution/__tests__/update-status.test.ts +58 -2
- package/src/domain/agentexecution/__tests__/validate-thinking-mode.test.ts +209 -61
- package/src/domain/agentexecution/controller.ts +7 -3
- package/src/domain/agentexecution/update-status.ts +43 -3
- package/src/domain/agentexecution/validate-thinking-mode.ts +102 -47
- package/src/domain/artifact/__tests__/store-faults.test.ts +117 -0
- package/src/domain/artifact/controller.ts +12 -6
- package/src/domain/environment/__tests__/store-faults.test.ts +150 -0
- package/src/domain/environment/controller.ts +8 -5
- package/src/domain/environment/steps.ts +10 -7
- package/src/domain/workflow/registry/data/model-registry.json +23 -14
- package/src/domain/workflow/registry/model-registry-store.ts +16 -3
- package/src/pipeline/__tests__/blind-not-found.test.ts +0 -3
|
@@ -1,29 +1,38 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* Pins validate-thinking-mode.ts
|
|
3
|
-
*
|
|
4
|
-
*
|
|
5
|
-
*
|
|
6
|
-
*
|
|
7
|
-
*
|
|
8
|
-
*
|
|
9
|
-
*
|
|
2
|
+
* Pins validate-thinking-mode.ts: fail-closed create-time validation of
|
|
3
|
+
* ExecutionConfig.thinking_mode (#772) against the BUNDLED registry, judged
|
|
4
|
+
* on the harness the execution will run on (#1280). The harness comes from
|
|
5
|
+
* the stored session for a turn on an existing session, else the bootstrap
|
|
6
|
+
* session_spec, UNSPECIFIED read as native; a real SQLite store holds the
|
|
7
|
+
* sessions. The bundled rows it leans on: claude-opus-4-6 declares
|
|
8
|
+
* `thinking` on its cursor entry and composer-2.5 declares none;
|
|
9
|
+
* claude-sonnet-5 is adaptive on its native entry and claude-haiku-4.5 is
|
|
10
|
+
* budget-shaped; claude-fable-5's native entry requires thinking.
|
|
10
11
|
*/
|
|
12
|
+
import { mkdtempSync, rmSync } from "node:fs";
|
|
13
|
+
import { tmpdir } from "node:os";
|
|
14
|
+
import path from "node:path";
|
|
15
|
+
|
|
11
16
|
import { testCallerIdentity } from "../../../pipeline/__tests__/support.js";
|
|
12
17
|
import { create } from "@bufbuild/protobuf";
|
|
13
18
|
import type { MessageInitShape } from "@bufbuild/protobuf";
|
|
14
19
|
import { Code, ConnectError } from "@connectrpc/connect";
|
|
15
|
-
import { describe, expect, it } from "vitest";
|
|
20
|
+
import { afterEach, beforeEach, describe, expect, it } from "vitest";
|
|
16
21
|
|
|
17
22
|
import { AgentExecutionSchema } from "@stigmer/protos/ai/stigmer/agentic/agentexecution/v1/api_pb";
|
|
18
23
|
import {
|
|
19
24
|
ServiceTier,
|
|
20
25
|
ThinkingMode,
|
|
21
26
|
} from "@stigmer/protos/ai/stigmer/agentic/agentexecution/v1/enum_pb";
|
|
22
|
-
import {
|
|
27
|
+
import { AgentExecutionSpecSchema } from "@stigmer/protos/ai/stigmer/agentic/agentexecution/v1/spec_pb";
|
|
28
|
+
import { SessionSchema } from "@stigmer/protos/ai/stigmer/agentic/session/v1/api_pb";
|
|
29
|
+
import { Harness } from "@stigmer/protos/ai/stigmer/agentic/session/v1/enum_pb";
|
|
23
30
|
import { ApiResourceKind } from "@stigmer/protos/ai/stigmer/commons/apiresource/apiresourcekind/api_resource_kind_pb";
|
|
24
31
|
|
|
25
32
|
import { createLogger } from "../../../boot/logger.js";
|
|
26
33
|
import { RequestContext } from "../../../pipeline/request-context.js";
|
|
34
|
+
import type { Store } from "../../../store/interface.js";
|
|
35
|
+
import { SqliteStore } from "../../../store/sqlite/store.js";
|
|
27
36
|
import { bundledModelRegistryDocument } from "../../workflow/registry/bundled.js";
|
|
28
37
|
import { ModelRegistryStore } from "../../workflow/registry/model-registry-store.js";
|
|
29
38
|
import { newValidateThinkingModeStep } from "../validate-thinking-mode.js";
|
|
@@ -41,16 +50,31 @@ const registry = new ModelRegistryStore({
|
|
|
41
50
|
logger: silentLogger,
|
|
42
51
|
});
|
|
43
52
|
|
|
44
|
-
|
|
53
|
+
let dir: string;
|
|
54
|
+
let store: Store;
|
|
55
|
+
|
|
56
|
+
beforeEach(() => {
|
|
57
|
+
dir = mkdtempSync(path.join(tmpdir(), "validate-thinking-mode-test-"));
|
|
58
|
+
store = SqliteStore.open(path.join(dir, "stigmer.db"));
|
|
59
|
+
});
|
|
60
|
+
|
|
61
|
+
afterEach(() => {
|
|
62
|
+
rmSync(dir, { recursive: true, force: true });
|
|
63
|
+
});
|
|
64
|
+
|
|
65
|
+
type SpecInit = MessageInitShape<typeof AgentExecutionSpecSchema>;
|
|
66
|
+
type ExecutionConfigInit = NonNullable<SpecInit["executionConfig"]>;
|
|
45
67
|
|
|
46
68
|
function contextFor(
|
|
47
|
-
config:
|
|
69
|
+
config: ExecutionConfigInit | undefined,
|
|
70
|
+
target: Pick<SpecInit, "sessionId" | "sessionSpec"> = {},
|
|
48
71
|
): RequestContext<typeof AgentExecutionSchema> {
|
|
49
72
|
return new RequestContext(
|
|
50
73
|
AgentExecutionSchema,
|
|
51
74
|
create(AgentExecutionSchema, {
|
|
52
75
|
spec: {
|
|
53
76
|
message: "hello",
|
|
77
|
+
...target,
|
|
54
78
|
...(config === undefined ? {} : { executionConfig: config }),
|
|
55
79
|
},
|
|
56
80
|
}),
|
|
@@ -59,11 +83,24 @@ function contextFor(
|
|
|
59
83
|
);
|
|
60
84
|
}
|
|
61
85
|
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
86
|
+
const onCursor = { sessionSpec: { harness: Harness.CURSOR } };
|
|
87
|
+
const onNative = { sessionSpec: { harness: Harness.NATIVE } };
|
|
88
|
+
|
|
89
|
+
async function passes(
|
|
90
|
+
config: ExecutionConfigInit | undefined,
|
|
91
|
+
target: Pick<SpecInit, "sessionId" | "sessionSpec"> = {},
|
|
92
|
+
onStore: Store = store,
|
|
93
|
+
): Promise<void> {
|
|
94
|
+
await newValidateThinkingModeStep(registry, onStore).execute(contextFor(config, target));
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
async function refusal(
|
|
98
|
+
config: ExecutionConfigInit,
|
|
99
|
+
target: Pick<SpecInit, "sessionId" | "sessionSpec"> = {},
|
|
100
|
+
onStore: Store = store,
|
|
101
|
+
): Promise<ConnectError> {
|
|
65
102
|
try {
|
|
66
|
-
|
|
103
|
+
await passes(config, target, onStore);
|
|
67
104
|
} catch (error) {
|
|
68
105
|
expect(error).toBeInstanceOf(ConnectError);
|
|
69
106
|
return error as ConnectError;
|
|
@@ -71,73 +108,184 @@ function refusal(
|
|
|
71
108
|
throw new Error("expected a fail-closed refusal, step passed");
|
|
72
109
|
}
|
|
73
110
|
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
111
|
+
async function storeSession(id: string, harness: Harness): Promise<void> {
|
|
112
|
+
await store.saveResource(
|
|
113
|
+
ApiResourceKind.session,
|
|
114
|
+
id,
|
|
115
|
+
SessionSchema,
|
|
116
|
+
create(SessionSchema, { metadata: { id, org: "acme" }, spec: { harness } }),
|
|
117
|
+
);
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
describe("ValidateThinkingMode: modes that need no harness", () => {
|
|
121
|
+
it("no execution_config passes", async () => {
|
|
122
|
+
await expect(passes(undefined)).resolves.toBeUndefined();
|
|
77
123
|
});
|
|
78
124
|
|
|
79
|
-
it("explicit DISABLED passes without a model", () => {
|
|
80
|
-
expect(()
|
|
81
|
-
step.execute(contextFor({ thinkingMode: ThinkingMode.DISABLED })),
|
|
82
|
-
).not.toThrow();
|
|
125
|
+
it("explicit DISABLED passes without a model", async () => {
|
|
126
|
+
await expect(passes({ thinkingMode: ThinkingMode.DISABLED })).resolves.toBeUndefined();
|
|
83
127
|
});
|
|
128
|
+
});
|
|
84
129
|
|
|
85
|
-
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
thinkingMode: ThinkingMode.ENABLED,
|
|
91
|
-
}),
|
|
92
|
-
),
|
|
93
|
-
).not.toThrow();
|
|
130
|
+
describe("ValidateThinkingMode on the cursor harness", () => {
|
|
131
|
+
it("ENABLED with a thinking-capable cursor model passes", async () => {
|
|
132
|
+
await expect(
|
|
133
|
+
passes({ modelName: "claude-opus-4-6", thinkingMode: ThinkingMode.ENABLED }, onCursor),
|
|
134
|
+
).resolves.toBeUndefined();
|
|
94
135
|
});
|
|
95
136
|
|
|
96
|
-
it("ENABLED combines freely with FAST — the combination bills as the fast variant", () => {
|
|
97
|
-
expect(
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
serviceTier: ServiceTier.FAST,
|
|
102
|
-
thinkingMode: ThinkingMode.ENABLED,
|
|
103
|
-
}),
|
|
137
|
+
it("ENABLED combines freely with FAST — the combination bills as the fast variant", async () => {
|
|
138
|
+
await expect(
|
|
139
|
+
passes(
|
|
140
|
+
{ modelName: "claude-opus-4-6", serviceTier: ServiceTier.FAST, thinkingMode: ThinkingMode.ENABLED },
|
|
141
|
+
onCursor,
|
|
104
142
|
),
|
|
105
|
-
).
|
|
143
|
+
).resolves.toBeUndefined();
|
|
106
144
|
});
|
|
107
145
|
|
|
108
|
-
it("ENABLED without model_name fails closed
|
|
109
|
-
const err = refusal({ thinkingMode: ThinkingMode.ENABLED });
|
|
146
|
+
it("ENABLED without model_name fails closed, naming the cursor models that think", async () => {
|
|
147
|
+
const err = await refusal({ thinkingMode: ThinkingMode.ENABLED }, onCursor);
|
|
110
148
|
expect(err.code).toBe(Code.InvalidArgument);
|
|
111
149
|
expect(err.rawMessage).toContain("requires execution_config.model_name");
|
|
112
|
-
// Thinking-capable suggestions ride along.
|
|
113
150
|
expect(err.rawMessage).toContain("claude-opus-4-6");
|
|
114
151
|
});
|
|
115
152
|
|
|
116
|
-
it("ENABLED on a cursor model without the capability fails closed", () => {
|
|
117
|
-
const err = refusal({
|
|
118
|
-
modelName: "composer-2.5",
|
|
119
|
-
thinkingMode: ThinkingMode.ENABLED,
|
|
120
|
-
});
|
|
153
|
+
it("ENABLED on a cursor model without the capability fails closed", async () => {
|
|
154
|
+
const err = await refusal({ modelName: "composer-2.5", thinkingMode: ThinkingMode.ENABLED }, onCursor);
|
|
121
155
|
expect(err.code).toBe(Code.InvalidArgument);
|
|
122
156
|
expect(err.rawMessage).toContain("no thinking capability");
|
|
123
157
|
expect(err.rawMessage).toContain("composer-2.5");
|
|
158
|
+
expect(err.rawMessage).toContain("cursor harness");
|
|
124
159
|
});
|
|
125
160
|
|
|
126
|
-
it("ENABLED on a native-only model fails closed
|
|
127
|
-
const err = refusal({
|
|
128
|
-
modelName: "claude-sonnet-4.6",
|
|
129
|
-
thinkingMode: ThinkingMode.ENABLED,
|
|
130
|
-
});
|
|
161
|
+
it("ENABLED on a native-only model fails closed on cursor: the cursor entry is the one that decides", async () => {
|
|
162
|
+
const err = await refusal({ modelName: "claude-sonnet-4.6", thinkingMode: ThinkingMode.ENABLED }, onCursor);
|
|
131
163
|
expect(err.code).toBe(Code.InvalidArgument);
|
|
132
|
-
expect(err.rawMessage).toContain("cursor");
|
|
133
|
-
expect(err.rawMessage).toContain("claude-sonnet-4.6");
|
|
164
|
+
expect(err.rawMessage).toContain("cursor harness");
|
|
134
165
|
});
|
|
135
166
|
|
|
136
|
-
it("
|
|
137
|
-
|
|
138
|
-
modelName: "
|
|
139
|
-
|
|
140
|
-
|
|
167
|
+
it("DISABLED on claude-fable-5 passes on cursor: only its native entry requires thinking", async () => {
|
|
168
|
+
await expect(
|
|
169
|
+
passes({ modelName: "claude-fable-5", thinkingMode: ThinkingMode.DISABLED }, onCursor),
|
|
170
|
+
).resolves.toBeUndefined();
|
|
171
|
+
});
|
|
172
|
+
});
|
|
173
|
+
|
|
174
|
+
describe("ValidateThinkingMode on the native harness (#1280)", () => {
|
|
175
|
+
it("ENABLED on an adaptive native model passes — the #1280 case, claude-sonnet-5 on a native session", async () => {
|
|
176
|
+
await expect(
|
|
177
|
+
passes({ modelName: "claude-sonnet-5", thinkingMode: ThinkingMode.ENABLED }, onNative),
|
|
178
|
+
).resolves.toBeUndefined();
|
|
179
|
+
});
|
|
180
|
+
|
|
181
|
+
it("ENABLED on a budget-shaped native model passes", async () => {
|
|
182
|
+
await expect(
|
|
183
|
+
passes({ modelName: "claude-haiku-4.5", thinkingMode: ThinkingMode.ENABLED }, onNative),
|
|
184
|
+
).resolves.toBeUndefined();
|
|
185
|
+
});
|
|
186
|
+
|
|
187
|
+
it("an UNSPECIFIED harness is native", async () => {
|
|
188
|
+
await expect(
|
|
189
|
+
passes({ modelName: "claude-sonnet-4.6", thinkingMode: ThinkingMode.ENABLED }, { sessionSpec: {} }),
|
|
190
|
+
).resolves.toBeUndefined();
|
|
191
|
+
await expect(passes({ modelName: "claude-sonnet-4.6", thinkingMode: ThinkingMode.ENABLED })).resolves.toBeUndefined();
|
|
192
|
+
});
|
|
193
|
+
|
|
194
|
+
it("ENABLED on a model with no native entry fails closed, naming the native models that think", async () => {
|
|
195
|
+
const err = await refusal({ modelName: "composer-2.5", thinkingMode: ThinkingMode.ENABLED }, onNative);
|
|
196
|
+
expect(err.code).toBe(Code.InvalidArgument);
|
|
197
|
+
expect(err.rawMessage).toContain("native harness");
|
|
198
|
+
expect(err.rawMessage).toContain("claude-sonnet-5");
|
|
199
|
+
expect(err.rawMessage).toContain("claude-haiku-4.5");
|
|
200
|
+
});
|
|
201
|
+
|
|
202
|
+
it("ENABLED on an unknown model fails closed", async () => {
|
|
203
|
+
const err = await refusal({ modelName: "not-a-model", thinkingMode: ThinkingMode.ENABLED }, onNative);
|
|
204
|
+
expect(err.code).toBe(Code.InvalidArgument);
|
|
205
|
+
});
|
|
206
|
+
|
|
207
|
+
it("an explicit DISABLED on a model that requires thinking fails closed", async () => {
|
|
208
|
+
const err = await refusal({ modelName: "claude-fable-5", thinkingMode: ThinkingMode.DISABLED }, onNative);
|
|
141
209
|
expect(err.code).toBe(Code.InvalidArgument);
|
|
210
|
+
expect(err.rawMessage).toContain("always thinks");
|
|
211
|
+
expect(err.rawMessage).toContain("claude-fable-5");
|
|
212
|
+
});
|
|
213
|
+
|
|
214
|
+
it("UNSPECIFIED and ENABLED pass on a model that requires thinking", async () => {
|
|
215
|
+
await expect(passes({ modelName: "claude-fable-5" }, onNative)).resolves.toBeUndefined();
|
|
216
|
+
await expect(
|
|
217
|
+
passes({ modelName: "claude-fable-5", thinkingMode: ThinkingMode.ENABLED }, onNative),
|
|
218
|
+
).resolves.toBeUndefined();
|
|
219
|
+
});
|
|
220
|
+
|
|
221
|
+
it("an explicit DISABLED passes on a model that may turn thinking off", async () => {
|
|
222
|
+
await expect(
|
|
223
|
+
passes({ modelName: "claude-sonnet-5", thinkingMode: ThinkingMode.DISABLED }, onNative),
|
|
224
|
+
).resolves.toBeUndefined();
|
|
225
|
+
});
|
|
226
|
+
});
|
|
227
|
+
|
|
228
|
+
describe("ValidateThinkingMode reads a stored session's harness", () => {
|
|
229
|
+
it("a turn on a stored cursor session is judged on cursor", async () => {
|
|
230
|
+
await storeSession("ses_cursor", Harness.CURSOR);
|
|
231
|
+
|
|
232
|
+
await expect(
|
|
233
|
+
passes({ modelName: "claude-opus-4-6", thinkingMode: ThinkingMode.ENABLED }, { sessionId: "ses_cursor" }),
|
|
234
|
+
).resolves.toBeUndefined();
|
|
235
|
+
const err = await refusal({ modelName: "claude-sonnet-4.6", thinkingMode: ThinkingMode.ENABLED }, { sessionId: "ses_cursor" });
|
|
236
|
+
expect(err.rawMessage).toContain("cursor harness");
|
|
237
|
+
});
|
|
238
|
+
|
|
239
|
+
it("a turn on a stored native session is judged on native, whatever the request's session_spec says", async () => {
|
|
240
|
+
await storeSession("ses_native", Harness.NATIVE);
|
|
241
|
+
|
|
242
|
+
await expect(
|
|
243
|
+
passes(
|
|
244
|
+
{ modelName: "claude-sonnet-5", thinkingMode: ThinkingMode.ENABLED },
|
|
245
|
+
{ sessionId: "ses_native", sessionSpec: { harness: Harness.CURSOR } },
|
|
246
|
+
),
|
|
247
|
+
).resolves.toBeUndefined();
|
|
248
|
+
});
|
|
249
|
+
|
|
250
|
+
it("a dangling session id is judged on the default native harness", async () => {
|
|
251
|
+
await expect(
|
|
252
|
+
passes({ modelName: "claude-sonnet-5", thinkingMode: ThinkingMode.ENABLED }, { sessionId: "ses_missing" }),
|
|
253
|
+
).resolves.toBeUndefined();
|
|
254
|
+
});
|
|
255
|
+
|
|
256
|
+
it("a store fault is Internal, never a reading of the session", async () => {
|
|
257
|
+
const faulty = new Proxy(store, {
|
|
258
|
+
get(target, prop, receiver) {
|
|
259
|
+
if (prop === "getResource") {
|
|
260
|
+
return async () => {
|
|
261
|
+
throw new Error("simulated store fault");
|
|
262
|
+
};
|
|
263
|
+
}
|
|
264
|
+
return Reflect.get(target, prop, receiver);
|
|
265
|
+
},
|
|
266
|
+
});
|
|
267
|
+
|
|
268
|
+
const err = await refusal(
|
|
269
|
+
{ modelName: "claude-sonnet-5", thinkingMode: ThinkingMode.ENABLED },
|
|
270
|
+
{ sessionId: "ses_any" },
|
|
271
|
+
faulty,
|
|
272
|
+
);
|
|
273
|
+
expect(err.code).toBe(Code.Internal);
|
|
274
|
+
});
|
|
275
|
+
|
|
276
|
+
it("a mode that needs no harness never reads the store", async () => {
|
|
277
|
+
const untouchable = new Proxy(store, {
|
|
278
|
+
get(target, prop, receiver) {
|
|
279
|
+
if (prop === "getResource") {
|
|
280
|
+
throw new Error("the store must not be read");
|
|
281
|
+
}
|
|
282
|
+
return Reflect.get(target, prop, receiver);
|
|
283
|
+
},
|
|
284
|
+
});
|
|
285
|
+
|
|
286
|
+
await expect(passes(undefined, { sessionId: "ses_any" }, untouchable)).resolves.toBeUndefined();
|
|
287
|
+
await expect(
|
|
288
|
+
passes({ thinkingMode: ThinkingMode.DISABLED }, { sessionId: "ses_any" }, untouchable),
|
|
289
|
+
).resolves.toBeUndefined();
|
|
142
290
|
});
|
|
143
291
|
});
|
|
@@ -279,12 +279,16 @@ export function registerAgentExecutionServices(
|
|
|
279
279
|
|
|
280
280
|
/**
|
|
281
281
|
* Create — create.go buildCreatePipeline, step-for-step: validation
|
|
282
|
-
* (proto → visibility → tier #357
|
|
282
|
+
* (proto → visibility → tier #357) → the run gate
|
|
283
283
|
* (AuthorizeRunTarget, P1 sp.run-gate: asking the target's own permission
|
|
284
284
|
* by request shape — session, instance or blueprint; the all-empty shape
|
|
285
285
|
* is the built-in assistant, admitted by Authorize's organization check
|
|
286
286
|
* and gated on nothing further — before the engine gate so a denied caller
|
|
287
|
-
* learns nothing about engine state, and before every side effect) → the
|
|
287
|
+
* learns nothing about engine state, and before every side effect) → the
|
|
288
|
+
* thinking-mode validation (#772), which judges the model on the harness
|
|
289
|
+
* the execution will run on and so may read the stored session: it runs
|
|
290
|
+
* behind the run gate, so nothing about a session is read or disclosed
|
|
291
|
+
* before the caller may add a turn to it (#1280) → the standard build
|
|
288
292
|
* → the engine gate (fail fast BEFORE the first side effect, so a down
|
|
289
293
|
* engine orphans nothing) → the pre-side-effect gate slot (O4; empty in OSS) → the
|
|
290
294
|
* side-effecting steps (default instance, session bootstrap,
|
|
@@ -317,10 +321,10 @@ async function createExecution(
|
|
|
317
321
|
.addStep(newValidateProtoStep())
|
|
318
322
|
.addStep(newValidateVisibilityStep())
|
|
319
323
|
.addStep(newValidateServiceTierStep(deps.modelRegistry))
|
|
320
|
-
.addStep(newValidateThinkingModeStep(deps.modelRegistry))
|
|
321
324
|
.addStep(
|
|
322
325
|
newAuthorizeRunTargetStep(deps.authorizer, agentExecutionRunTarget),
|
|
323
326
|
)
|
|
327
|
+
.addStep(newValidateThinkingModeStep(deps.modelRegistry, deps.store))
|
|
324
328
|
.addStep(newResolveSlugStep())
|
|
325
329
|
.addStep(newBuildNewStateStep())
|
|
326
330
|
// Vouches the runner-stamped workflow lineage labels (or refuses a
|
|
@@ -29,7 +29,11 @@
|
|
|
29
29
|
* appended and never replaces the transcript, and a PAUSED or CANCELLED
|
|
30
30
|
* execution's stop row stays last across the runner's later writes,
|
|
31
31
|
* because the Pause and Cancel RPCs, the invoke workflow and the runner
|
|
32
|
-
* all write a stopped execution and nothing orders them.
|
|
32
|
+
* all write a stopped execution and nothing orders them. For the same
|
|
33
|
+
* reason a PAUSED execution's phase moves only on the platform's own
|
|
34
|
+
* write (stigmer#1370): the runner learns of a pause on a later heartbeat
|
|
35
|
+
* and keeps streaming IN_PROGRESS until then, and without the latch its
|
|
36
|
+
* stragglers un-pause the execution and Resume finds nothing to resume.
|
|
33
37
|
*
|
|
34
38
|
* Broadcast rides in-memory channels (ADR 011).
|
|
35
39
|
*/
|
|
@@ -66,6 +70,7 @@ import {
|
|
|
66
70
|
} from "../../pipeline/errors.js";
|
|
67
71
|
import { newPipeline } from "../../pipeline/pipeline.js";
|
|
68
72
|
import type { CallerIdentity } from "../../extensions/identity.js";
|
|
73
|
+
import { isServerComposedRequest } from "../../extensions/identity.js";
|
|
69
74
|
import { RequestContext } from "../../pipeline/request-context.js";
|
|
70
75
|
import { newAuthorizeStep } from "../../pipeline/steps/authorize.js";
|
|
71
76
|
import { ResourceNotFoundError } from "../../store/interface.js";
|
|
@@ -161,7 +166,12 @@ export async function updateStatus(
|
|
|
161
166
|
oldPhase =
|
|
162
167
|
execution.status?.phase ??
|
|
163
168
|
ExecutionPhase.EXECUTION_PHASE_UNSPECIFIED;
|
|
164
|
-
applyUpdateStatusMerge(
|
|
169
|
+
applyUpdateStatusMerge(
|
|
170
|
+
execution,
|
|
171
|
+
ctx.input,
|
|
172
|
+
deps.logger,
|
|
173
|
+
isServerComposedRequest(identity) ? "platform" : "wire",
|
|
174
|
+
);
|
|
165
175
|
},
|
|
166
176
|
);
|
|
167
177
|
} catch (error) {
|
|
@@ -315,18 +325,27 @@ function mergeTranscript(
|
|
|
315
325
|
return next;
|
|
316
326
|
}
|
|
317
327
|
|
|
328
|
+
/**
|
|
329
|
+
* Who sent a status write: the platform's own (the invoke workflow's
|
|
330
|
+
* writes, which reach this handler over the in-process transport, the
|
|
331
|
+
* one origin the wire cannot claim) or the wire (the runner).
|
|
332
|
+
*/
|
|
333
|
+
export type StatusWriter = "platform" | "wire";
|
|
334
|
+
|
|
318
335
|
/**
|
|
319
336
|
* Merges an incoming status update into the execution in place — the
|
|
320
337
|
* runner-owns-the-transcript merge rules, exactly update_status.go
|
|
321
338
|
* applyUpdateStatusMerge. Runs inside the updateResource closure
|
|
322
339
|
* (synchronous by store contract), so the merge, the approval event
|
|
323
340
|
* authoring, and the pending_approvals projection all see the same
|
|
324
|
-
* snapshot that will be persisted.
|
|
341
|
+
* snapshot that will be persisted. `writer` defaults to the wire, the
|
|
342
|
+
* side that grants nothing extra.
|
|
325
343
|
*/
|
|
326
344
|
export function applyUpdateStatusMerge(
|
|
327
345
|
execution: AgentExecution,
|
|
328
346
|
input: AgentExecutionUpdateStatusInput,
|
|
329
347
|
logger: Logger,
|
|
348
|
+
writer: StatusWriter = "wire",
|
|
330
349
|
): void {
|
|
331
350
|
if (execution.status === undefined) {
|
|
332
351
|
execution.status = create(AgentExecutionStatusSchema);
|
|
@@ -415,8 +434,29 @@ export function applyUpdateStatusMerge(
|
|
|
415
434
|
// re-terminalize it is gone. Recover is the one sanctioned
|
|
416
435
|
// un-terminalizer and runs through its own lifecycle step, never this
|
|
417
436
|
// merge.
|
|
437
|
+
//
|
|
438
|
+
// A PAUSED execution is latched the same way against the wire
|
|
439
|
+
// (stigmer#1370): the runner's straggler persists, sent before its
|
|
440
|
+
// heartbeat delivers the pause, carry IN_PROGRESS. Resume moves the
|
|
441
|
+
// phase through its own lifecycle step; through this merge only the
|
|
442
|
+
// platform's own writes do (the workflow's resume and recovery
|
|
443
|
+
// re-assertions, which also heal a stale PAUSED over a fast resume,
|
|
444
|
+
// oss#869). The straggler's transcript still merges, beneath the held
|
|
445
|
+
// pause row.
|
|
418
446
|
if (requestStatus.phase !== ExecutionPhase.EXECUTION_PHASE_UNSPECIFIED) {
|
|
419
447
|
if (
|
|
448
|
+
existingPhase === ExecutionPhase.EXECUTION_PAUSED &&
|
|
449
|
+
writer === "wire" &&
|
|
450
|
+
requestStatus.phase !== existingPhase
|
|
451
|
+
) {
|
|
452
|
+
logger.warn(
|
|
453
|
+
"Ignored a wire phase change on a paused execution; Resume or the platform's own write moves it",
|
|
454
|
+
{
|
|
455
|
+
executionId: input.executionId,
|
|
456
|
+
ignoredPhase: ExecutionPhase[requestStatus.phase],
|
|
457
|
+
},
|
|
458
|
+
);
|
|
459
|
+
} else if (
|
|
420
460
|
isTerminalExecutionPhase(existingPhase) &&
|
|
421
461
|
requestStatus.phase !== existingPhase
|
|
422
462
|
) {
|
|
@@ -1,88 +1,143 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* ValidateThinkingMode —
|
|
3
|
-
*
|
|
4
|
-
* (stigmer/stigmer#772)
|
|
5
|
-
*
|
|
2
|
+
* ValidateThinkingMode — fail-closed validation of
|
|
3
|
+
* ExecutionConfig.thinking_mode against the model registry
|
|
4
|
+
* (stigmer/stigmer#772), the sibling of ValidateServiceTier for the second
|
|
5
|
+
* variant dimension, judged on the harness the execution will run on
|
|
6
|
+
* (stigmer/stigmer#1280).
|
|
6
7
|
*
|
|
7
8
|
* Unlike the fast tier, thinking is capability-gated, not pricing-gated:
|
|
8
|
-
* thinking variants bill at base per-token rates (ledger-verified), so
|
|
9
|
-
*
|
|
10
|
-
*
|
|
11
|
-
*
|
|
12
|
-
*
|
|
13
|
-
*
|
|
14
|
-
*
|
|
15
|
-
*
|
|
9
|
+
* thinking variants bill at base per-token rates (ledger-verified), so the
|
|
10
|
+
* registry fact that makes ENABLED selectable is a thinking capability on
|
|
11
|
+
* the model's entry for the execution's harness — `thinking` (a fixed
|
|
12
|
+
* budget) or `adaptiveThinking` (depth adapts). Both harnesses translate
|
|
13
|
+
* the mode: Cursor as its thinking variant, native as Anthropic's
|
|
14
|
+
* `thinking` parameter in the form the entry declares. The harness scoping
|
|
15
|
+
* is what keeps a selection honest: an id such as claude-sonnet-5 has an
|
|
16
|
+
* entry on each harness, and only the entry the execution runs on says
|
|
17
|
+
* what its model does.
|
|
16
18
|
*
|
|
17
|
-
* - UNSPECIFIED
|
|
18
|
-
*
|
|
19
|
-
*
|
|
19
|
+
* - UNSPECIFIED: always valid — it resolves to DISABLED in the runner,
|
|
20
|
+
* and a model that requires thinking thinks under it.
|
|
21
|
+
* - DISABLED: valid, except on a model whose entry declares
|
|
22
|
+
* `thinkingRequired` — it always thinks and the provider refuses a
|
|
23
|
+
* request to turn thinking off.
|
|
20
24
|
* - ENABLED: requires model_name to be set (Auto has no variant
|
|
21
|
-
* dimensions) and that model's
|
|
22
|
-
*
|
|
25
|
+
* dimensions) and that model's entry on the harness to declare a
|
|
26
|
+
* thinking capability.
|
|
23
27
|
*
|
|
24
|
-
*
|
|
25
|
-
*
|
|
26
|
-
*
|
|
28
|
+
* The harness is the one CreateSessionIfNeeded and dispatch will use: the
|
|
29
|
+
* stored session's for a turn on an existing session, else the bootstrap
|
|
30
|
+
* session_spec's, UNSPECIFIED read as native (harnessName; the Cloud
|
|
31
|
+
* billing gate resolves the same way). Reading a stored session is why the
|
|
32
|
+
* step runs after the run gate: the caller is authorized to add a turn to
|
|
33
|
+
* that session before anything about it is read, so a refusal never tells
|
|
34
|
+
* an unauthorized caller which harness a session uses. It is still a pure
|
|
35
|
+
* step, before any side effect.
|
|
27
36
|
*/
|
|
28
|
-
import type { AgentExecutionSchema } from "@stigmer/protos/ai/stigmer/agentic/agentexecution/v1/api_pb";
|
|
37
|
+
import type { AgentExecution, AgentExecutionSchema } from "@stigmer/protos/ai/stigmer/agentic/agentexecution/v1/api_pb";
|
|
29
38
|
import { ThinkingMode } from "@stigmer/protos/ai/stigmer/agentic/agentexecution/v1/enum_pb";
|
|
39
|
+
import { SessionSchema } from "@stigmer/protos/ai/stigmer/agentic/session/v1/api_pb";
|
|
40
|
+
import { Harness } from "@stigmer/protos/ai/stigmer/agentic/session/v1/enum_pb";
|
|
41
|
+
import { ApiResourceKind } from "@stigmer/protos/ai/stigmer/commons/apiresource/apiresourcekind/api_resource_kind_pb";
|
|
30
42
|
|
|
31
|
-
import { invalidArgumentError } from "../../pipeline/errors.js";
|
|
43
|
+
import { internalError, invalidArgumentError } from "../../pipeline/errors.js";
|
|
32
44
|
import type { PipelineStep } from "../../pipeline/pipeline.js";
|
|
45
|
+
import { ResourceNotFoundError, type Store } from "../../store/interface.js";
|
|
33
46
|
import type { ModelCatalogProvider } from "../workflow/registry/model-catalog-provider.js";
|
|
34
|
-
import {
|
|
35
|
-
|
|
47
|
+
import {
|
|
48
|
+
ADAPTIVE_THINKING_CAPABILITY_KEY,
|
|
49
|
+
THINKING_CAPABILITY_KEY,
|
|
50
|
+
THINKING_REQUIRED_CAPABILITY_KEY,
|
|
51
|
+
} from "../workflow/registry/model-registry-store.js";
|
|
52
|
+
import { harnessName } from "../workflow/registry/pin-validation.js";
|
|
36
53
|
|
|
37
54
|
export function newValidateThinkingModeStep(
|
|
38
55
|
registry: ModelCatalogProvider,
|
|
56
|
+
store: Store,
|
|
39
57
|
): PipelineStep<typeof AgentExecutionSchema> {
|
|
40
58
|
return {
|
|
41
59
|
name: "ValidateThinkingMode",
|
|
42
|
-
execute(ctx) {
|
|
60
|
+
async execute(ctx) {
|
|
43
61
|
const config = ctx.newState.spec?.executionConfig;
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
62
|
+
const mode = config?.thinkingMode ?? ThinkingMode.UNSPECIFIED;
|
|
63
|
+
const modelName = (config?.modelName ?? "").trim();
|
|
64
|
+
// UNSPECIFIED is always valid, and so is DISABLED on Auto (no model
|
|
65
|
+
// to require thinking). Unknown enum numbers were already refused by
|
|
66
|
+
// proto field validation (defined_only).
|
|
67
|
+
if (mode === ThinkingMode.UNSPECIFIED || (mode === ThinkingMode.DISABLED && modelName === "")) {
|
|
68
|
+
return;
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
const harness = await executionHarness(ctx.newState, store);
|
|
72
|
+
|
|
73
|
+
if (mode === ThinkingMode.DISABLED) {
|
|
74
|
+
if (registry.hasCapabilityForHarness(harness, modelName, THINKING_REQUIRED_CAPABILITY_KEY)) {
|
|
75
|
+
throw invalidArgumentError(
|
|
76
|
+
`thinking_mode 'disabled' is not available for model '${modelName}': the model always ` +
|
|
77
|
+
`thinks on the ${harness} harness and refuses a request to turn thinking off. ` +
|
|
78
|
+
"Leave thinking_mode unset, or set it to 'enabled'.",
|
|
79
|
+
);
|
|
80
|
+
}
|
|
47
81
|
return;
|
|
48
82
|
}
|
|
49
83
|
|
|
50
|
-
const modelName = (config.modelName ?? "").trim();
|
|
51
84
|
if (modelName === "") {
|
|
52
85
|
throw invalidArgumentError(
|
|
53
86
|
"thinking_mode 'enabled' requires execution_config.model_name: thinking is a " +
|
|
54
87
|
"per-model capability, and Auto (no pinned model) has no variant dimensions. " +
|
|
55
|
-
`Pin a model that supports it${thinkingCapableSuffix(registry)}.`,
|
|
88
|
+
`Pin a model that supports it${thinkingCapableSuffix(registry, harness)}.`,
|
|
56
89
|
);
|
|
57
90
|
}
|
|
58
91
|
|
|
59
|
-
if (
|
|
60
|
-
!registry.hasCapabilityForHarness(
|
|
61
|
-
HARNESS_NAME_CURSOR,
|
|
62
|
-
modelName,
|
|
63
|
-
THINKING_CAPABILITY_KEY,
|
|
64
|
-
)
|
|
65
|
-
) {
|
|
92
|
+
if (!canThink(registry, harness, modelName)) {
|
|
66
93
|
throw invalidArgumentError(
|
|
67
94
|
`thinking_mode 'enabled' is not available for model '${modelName}': the model ` +
|
|
68
|
-
|
|
69
|
-
`harness${thinkingCapableSuffix(registry)}.`,
|
|
95
|
+
`registry declares no thinking capability for it on the ${harness} ` +
|
|
96
|
+
`harness${thinkingCapableSuffix(registry, harness)}.`,
|
|
70
97
|
);
|
|
71
98
|
}
|
|
72
99
|
},
|
|
73
100
|
};
|
|
74
101
|
}
|
|
75
102
|
|
|
103
|
+
/** The registry section name of the harness this execution will run on. */
|
|
104
|
+
async function executionHarness(execution: AgentExecution, store: Store): Promise<string> {
|
|
105
|
+
const sessionId = execution.spec?.sessionId ?? "";
|
|
106
|
+
if (sessionId === "") {
|
|
107
|
+
return harnessName(execution.spec?.sessionSpec?.harness ?? Harness.UNSPECIFIED);
|
|
108
|
+
}
|
|
109
|
+
try {
|
|
110
|
+
const session = await store.getResource(ApiResourceKind.session, sessionId, SessionSchema);
|
|
111
|
+
return harnessName(session.spec?.harness ?? Harness.UNSPECIFIED);
|
|
112
|
+
} catch (error) {
|
|
113
|
+
// A dangling session id is the loading steps' refusal to make, with
|
|
114
|
+
// its own NotFound; judge it on the default harness, as dispatch does.
|
|
115
|
+
// Any other failure is a store fault, never a reading of the session.
|
|
116
|
+
if (error instanceof ResourceNotFoundError) {
|
|
117
|
+
return harnessName(Harness.UNSPECIFIED);
|
|
118
|
+
}
|
|
119
|
+
throw internalError(error, "failed to load session for thinking-mode validation");
|
|
120
|
+
}
|
|
121
|
+
}
|
|
122
|
+
|
|
123
|
+
function canThink(registry: ModelCatalogProvider, harness: string, modelName: string): boolean {
|
|
124
|
+
return (
|
|
125
|
+
registry.hasCapabilityForHarness(harness, modelName, THINKING_CAPABILITY_KEY) ||
|
|
126
|
+
registry.hasCapabilityForHarness(harness, modelName, ADAPTIVE_THINKING_CAPABILITY_KEY)
|
|
127
|
+
);
|
|
128
|
+
}
|
|
129
|
+
|
|
76
130
|
/**
|
|
77
|
-
* "; models with a thinking mode: a, b, c" — actionable refusal detail
|
|
78
|
-
*
|
|
79
|
-
* declares none.
|
|
131
|
+
* "; models with a thinking mode: a, b, c" — actionable refusal detail for
|
|
132
|
+
* the harness, sorted, empty when the registry declares none there.
|
|
80
133
|
*/
|
|
81
|
-
function thinkingCapableSuffix(registry: ModelCatalogProvider): string {
|
|
82
|
-
const capable =
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
134
|
+
function thinkingCapableSuffix(registry: ModelCatalogProvider, harness: string): string {
|
|
135
|
+
const capable = [
|
|
136
|
+
...new Set([
|
|
137
|
+
...registry.canonicalModelsWithCapabilityForHarness(harness, THINKING_CAPABILITY_KEY),
|
|
138
|
+
...registry.canonicalModelsWithCapabilityForHarness(harness, ADAPTIVE_THINKING_CAPABILITY_KEY),
|
|
139
|
+
]),
|
|
140
|
+
].sort();
|
|
86
141
|
if (capable.length === 0) {
|
|
87
142
|
return "";
|
|
88
143
|
}
|