@gajae-code/ai 0.12.4 → 0.12.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +12 -0
- package/dist/types/providers/dashscope-token-plan-headers.d.ts +57 -0
- package/dist/types/types.d.ts +21 -3
- package/dist/types/utils/provider-response.d.ts +5 -2
- package/package.json +2 -2
- package/src/providers/amazon-bedrock.ts +1 -1
- package/src/providers/anthropic.ts +5 -3
- package/src/providers/azure-openai-responses.ts +4 -2
- package/src/providers/cursor.ts +1 -1
- package/src/providers/dashscope-token-plan-headers.ts +84 -0
- package/src/providers/gitlab-duo.ts +3 -0
- package/src/providers/google-gemini-cli.ts +7 -2
- package/src/providers/google-shared.ts +6 -2
- package/src/providers/mock.ts +1 -0
- package/src/providers/ollama.ts +1 -1
- package/src/providers/openai-anthropic-shim.ts +2 -0
- package/src/providers/openai-codex-responses.ts +9 -3
- package/src/providers/openai-completions.ts +13 -2
- package/src/providers/openai-responses.ts +23 -4
- package/src/stream.ts +1 -0
- package/src/types.ts +30 -3
- package/src/utils/provider-response.ts +3 -3
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,18 @@
|
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
4
|
|
|
5
|
+
## [0.12.6] - 2026-07-31
|
|
6
|
+
|
|
7
|
+
## [0.12.5] - 2026-07-30
|
|
8
|
+
### Fixed
|
|
9
|
+
|
|
10
|
+
- Alibaba Token Plan requests now carry Qwen Code's canonical DashScope request fingerprint on both transports. The built-in `alibaba-token-plan` provider (openai-responses `qwen3.8-max-preview` and openai-completions `glm-5.2`/`deepseek-v4-pro`) now emits the four upstream identity/cache/auth headers (`User-Agent`, `X-DashScope-CacheControl: enable`, `X-DashScope-UserAgent`, `X-DashScope-AuthType: openai`) matching `QwenLM/qwen-code` v0.21.1 (commit `f4cd6e1`) exactly, via a shared helper. DashScope is compatibility-sensitive to this client fingerprint, so a non-identical set can cause request instability and affect first-event latency. Caller headers still win per key (upstream `{...default, ...customHeaders}` precedence); non-Alibaba providers are byte-unchanged (#3557).
|
|
11
|
+
|
|
12
|
+
### Added
|
|
13
|
+
|
|
14
|
+
- Reproducible Alibaba Token Plan header-parity A/B latency benchmark (`packages/ai/scripts/alibaba-token-plan-latency-ab.ts`): a fixed-seed interleaved A/B comparison of legacy vs Qwen-identical headers against a deterministic local HTTP server, reporting n/success/error/timeout and TTFT/total latency median/p90/p95/mean/stddev. No live credentials are required; a public-safe blocked-live-data receipt is included (`packages/ai/test/fixtures/alibaba-token-plan-latency-blocked-receipt.md`) (#3557).
|
|
15
|
+
|
|
16
|
+
|
|
5
17
|
## [0.12.4] - 2026-07-30
|
|
6
18
|
|
|
7
19
|
### Fixed
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* DashScope Token Plan canonical request headers.
|
|
3
|
+
*
|
|
4
|
+
* Reproduces QwenLM/qwen-code's DashScopeOpenAICompatibleProvider.buildHeaders()
|
|
5
|
+
* defaultHeaders so the built-in `alibaba-token-plan` provider emits the same
|
|
6
|
+
* client identity / cache / auth-type fingerprint upstream sends. DashScope is
|
|
7
|
+
* compatibility-sensitive to this fingerprint; a non-identical set can cause
|
|
8
|
+
* request instability and affect first-event latency (gajae-code #3557).
|
|
9
|
+
*
|
|
10
|
+
* Upstream pin (reproduce EXACTLY here):
|
|
11
|
+
* Repository: QwenLM/qwen-code
|
|
12
|
+
* Commit: f4cd6e1d8bbb1c24e7e5d1a40187d8e28aa7c4fb
|
|
13
|
+
* Version: 0.21.1
|
|
14
|
+
* Source: packages/core/src/core/openaiContentGenerator/provider/dashscope.ts
|
|
15
|
+
* buildHeaders():
|
|
16
|
+
* const userAgent = `QwenCode/${version} (${process.platform}; ${process.arch})`;
|
|
17
|
+
* const defaultHeaders = {
|
|
18
|
+
* 'User-Agent': userAgent,
|
|
19
|
+
* 'X-DashScope-CacheControl': 'enable',
|
|
20
|
+
* 'X-DashScope-UserAgent': userAgent,
|
|
21
|
+
* 'X-DashScope-AuthType': authType,
|
|
22
|
+
* };
|
|
23
|
+
* return customHeaders ? { ...defaultHeaders, ...customHeaders } : defaultHeaders;
|
|
24
|
+
*
|
|
25
|
+
* The Token Plan preset authenticates with AuthType.USE_OPENAI ('openai'), so
|
|
26
|
+
* X-DashScope-AuthType is the constant 'openai'. Pin the version so an upstream
|
|
27
|
+
* bump is an explicit parity update rather than silent drift.
|
|
28
|
+
*/
|
|
29
|
+
export declare const QWEN_CODE_UPSTREAM_REPO = "QwenLM/qwen-code";
|
|
30
|
+
export declare const QWEN_CODE_UPSTREAM_COMMIT = "f4cd6e1d8bbb1c24e7e5d1a40187d8e28aa7c4fb";
|
|
31
|
+
export declare const QWEN_CODE_UPSTREAM_VERSION = "0.21.1";
|
|
32
|
+
/**
|
|
33
|
+
* The Qwen Code CLI version string used in identity headers. Pinned to the
|
|
34
|
+
* upstream version at {@link QWEN_CODE_UPSTREAM_COMMIT}; change both together
|
|
35
|
+
* as an explicit parity update.
|
|
36
|
+
*/
|
|
37
|
+
export declare function qwenCodeUserAgent(version?: string): string;
|
|
38
|
+
/**
|
|
39
|
+
* Canonical DashScope Token Plan headers (upstream defaultHeaders, no caller
|
|
40
|
+
* overrides applied). Exposed for tests/fixtures so the pinned wire set lives
|
|
41
|
+
* in exactly one place.
|
|
42
|
+
*/
|
|
43
|
+
export declare function dashscopeTokenPlanDefaultHeaders(version?: string): Readonly<Record<string, string>>;
|
|
44
|
+
/**
|
|
45
|
+
* Merge canonical DashScope Token Plan identity headers onto a caller's header
|
|
46
|
+
* map, reproducing upstream buildHeaders() precedence EXACTLY:
|
|
47
|
+
* `{ ...defaultHeaders, ...customHeaders }` — caller wins per header.
|
|
48
|
+
*
|
|
49
|
+
* This mirrors GJC's existing kimi-code injection order
|
|
50
|
+
* (`headers = { ...getKimiCommonHeaders(), ...headers }`): canonical identity as
|
|
51
|
+
* the base, caller-supplied headers overriding individual keys. A caller that
|
|
52
|
+
* pins `User-Agent` takes that key; the other canonicals still apply.
|
|
53
|
+
*
|
|
54
|
+
* A null/undefined `callerHeaders` returns the canonical set alone (upstream
|
|
55
|
+
* `customHeaders ? {...} : defaultHeaders` shortcut).
|
|
56
|
+
*/
|
|
57
|
+
export declare function mergeDashScopeTokenPlanHeaders(callerHeaders: Record<string, string> | undefined, version?: string): Record<string, string>;
|
package/dist/types/types.d.ts
CHANGED
|
@@ -224,19 +224,21 @@ export interface StreamOptions {
|
|
|
224
224
|
/**
|
|
225
225
|
* Optional callback for inspecting or replacing provider payloads before sending.
|
|
226
226
|
* Return undefined to keep the payload unchanged.
|
|
227
|
+
* The `scope` parameter carries the per-attempt identity for execution attribution.
|
|
227
228
|
*/
|
|
228
|
-
onPayload?: (payload: unknown, model?: Model<Api
|
|
229
|
+
onPayload?: (payload: unknown, model?: Model<Api>, scope?: AttemptScopeRef) => unknown | undefined | Promise<unknown | undefined>;
|
|
229
230
|
/**
|
|
230
231
|
* Optional callback for provider response metadata after headers are received.
|
|
232
|
+
* The `scope` parameter carries the per-attempt identity for execution attribution.
|
|
231
233
|
*/
|
|
232
|
-
onResponse?: (response: ProviderResponseMetadata, model?: Model<Api
|
|
234
|
+
onResponse?: (response: ProviderResponseMetadata, model?: Model<Api>, scope?: AttemptScopeRef) => void | Promise<void>;
|
|
233
235
|
/**
|
|
234
236
|
* Optional callback for raw Server-Sent Events as they arrive from HTTP streaming providers.
|
|
235
237
|
*
|
|
236
238
|
* Diagnostic only: provider implementations must ignore callback failures and must not
|
|
237
239
|
* let observers alter stream contents.
|
|
238
240
|
*/
|
|
239
|
-
onSseEvent?: (event: RawSseEvent, model?: Model<Api
|
|
241
|
+
onSseEvent?: (event: RawSseEvent, model?: Model<Api>, scope?: AttemptScopeRef) => void;
|
|
240
242
|
/**
|
|
241
243
|
* Optional override for the first streamed event watchdog in milliseconds.
|
|
242
244
|
* Set to 0 to disable the first-event watchdog for this request.
|
|
@@ -266,6 +268,22 @@ export interface StreamOptions {
|
|
|
266
268
|
authCredentialType?: "api_key" | "oauth";
|
|
267
269
|
/** Cursor exec/MCP tool handlers (cursor-agent only). */
|
|
268
270
|
execHandlers?: CursorExecHandlers;
|
|
271
|
+
/** Per-attempt identity for execution attribution. Threaded into onPayload/onResponse calls. */
|
|
272
|
+
attemptScope?: AttemptScopeRef;
|
|
273
|
+
}
|
|
274
|
+
/**
|
|
275
|
+
* Low-level structural carrier for per-attempt identity attribution.
|
|
276
|
+
*
|
|
277
|
+
* Defined in `packages/ai` so that {@link SimpleStreamOptions} and provider
|
|
278
|
+
* hook signatures can carry an attempt identity without a reverse dependency
|
|
279
|
+
* on `packages/agent`. The concrete `AttemptScope` in `packages/agent` is
|
|
280
|
+
* structurally assignable to this interface (same `attemptId` + `generation`
|
|
281
|
+
* + `lineage` fields).
|
|
282
|
+
*/
|
|
283
|
+
export interface AttemptScopeRef {
|
|
284
|
+
readonly attemptId: string;
|
|
285
|
+
readonly generation: number;
|
|
286
|
+
readonly lineage: string;
|
|
269
287
|
}
|
|
270
288
|
export interface SimpleStreamOptions extends StreamOptions {
|
|
271
289
|
reasoning?: Effort;
|
|
@@ -1,3 +1,6 @@
|
|
|
1
|
-
import type { Api, Model, ProviderResponseMetadata, StreamOptions } from "../types";
|
|
1
|
+
import type { Api, AttemptScopeRef, Model, ProviderResponseMetadata, StreamOptions } from "../types";
|
|
2
2
|
export declare function normalizeProviderResponse(response: Response, requestId?: string | null, metadata?: Record<string, unknown>): ProviderResponseMetadata;
|
|
3
|
-
export declare function notifyProviderResponse(options:
|
|
3
|
+
export declare function notifyProviderResponse(options: {
|
|
4
|
+
onResponse?: StreamOptions["onResponse"];
|
|
5
|
+
attemptScope?: AttemptScopeRef;
|
|
6
|
+
} | undefined, response: Response, model?: Model<Api>, requestId?: string | null, metadata?: Record<string, unknown>): Promise<void>;
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"type": "module",
|
|
3
3
|
"name": "@gajae-code/ai",
|
|
4
|
-
"version": "0.12.
|
|
4
|
+
"version": "0.12.6",
|
|
5
5
|
"description": "Unified LLM API with automatic model discovery and provider configuration",
|
|
6
6
|
"homepage": "https://gajae-code.com",
|
|
7
7
|
"author": "Yeachan-Heo and Gajae Code Contributors",
|
|
@@ -40,7 +40,7 @@
|
|
|
40
40
|
"dependencies": {
|
|
41
41
|
"@anthropic-ai/sdk": "^0.94.0",
|
|
42
42
|
"@bufbuild/protobuf": "^2.12.0",
|
|
43
|
-
"@gajae-code/utils": "0.12.
|
|
43
|
+
"@gajae-code/utils": "0.12.6",
|
|
44
44
|
"openai": "^6.36.0",
|
|
45
45
|
"partial-json": "^0.1.7",
|
|
46
46
|
"zod": "4.4.3"
|
|
@@ -228,7 +228,7 @@ export const streamBedrock: StreamFunction<"bedrock-converse-stream"> = (
|
|
|
228
228
|
toolConfig,
|
|
229
229
|
additionalModelRequestFields,
|
|
230
230
|
};
|
|
231
|
-
options?.onPayload?.(commandInput);
|
|
231
|
+
options?.onPayload?.(commandInput, model, options?.attemptScope);
|
|
232
232
|
|
|
233
233
|
const host = `bedrock-runtime.${region}.amazonaws.com`;
|
|
234
234
|
const url = `https://${host}/model/${encodeURIComponent(model.id)}/converse-stream`;
|
|
@@ -1348,7 +1348,9 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1348
1348
|
dynamicHeaders: copilotDynamicHeaders?.headers,
|
|
1349
1349
|
isOAuth: options?.isOAuth,
|
|
1350
1350
|
hasTools: !!context.tools?.length,
|
|
1351
|
-
onSseEvent: options?.onSseEvent
|
|
1351
|
+
onSseEvent: options?.onSseEvent
|
|
1352
|
+
? event => options.onSseEvent!(event, model, options?.attemptScope)
|
|
1353
|
+
: undefined,
|
|
1352
1354
|
fetch: options?.fetch,
|
|
1353
1355
|
requestMaxRetries: options?.requestMaxRetries,
|
|
1354
1356
|
maxRetryDelayMs: options?.maxRetryDelayMs,
|
|
@@ -1386,7 +1388,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1386
1388
|
if (dropFastMode) {
|
|
1387
1389
|
dropAnthropicFastMode(nextParams);
|
|
1388
1390
|
}
|
|
1389
|
-
const replacementPayload = await options?.onPayload?.(nextParams, model);
|
|
1391
|
+
const replacementPayload = await options?.onPayload?.(nextParams, model, options?.attemptScope);
|
|
1390
1392
|
if (replacementPayload !== undefined) {
|
|
1391
1393
|
nextParams = replacementPayload as typeof nextParams;
|
|
1392
1394
|
}
|
|
@@ -1489,7 +1491,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
|
|
|
1489
1491
|
} = await getAnthropicStreamResponse(
|
|
1490
1492
|
anthropicRequest,
|
|
1491
1493
|
requestSignal,
|
|
1492
|
-
options?.client ? event => options?.onSseEvent?.(event, model) : undefined,
|
|
1494
|
+
options?.client ? event => options?.onSseEvent?.(event, model, options?.attemptScope) : undefined,
|
|
1493
1495
|
);
|
|
1494
1496
|
await notifyProviderResponse(options, response, model, requestId);
|
|
1495
1497
|
let sawEvent = false;
|
|
@@ -126,7 +126,7 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
|
|
|
126
126
|
const { baseUrl } = resolveAzureConfig(model, options);
|
|
127
127
|
const params = buildParams(model, context, options, deploymentName, baseUrl);
|
|
128
128
|
const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs();
|
|
129
|
-
options?.onPayload?.(params);
|
|
129
|
+
options?.onPayload?.(params, model, options?.attemptScope);
|
|
130
130
|
rawRequestDump = {
|
|
131
131
|
provider: model.provider,
|
|
132
132
|
api: output.api,
|
|
@@ -312,7 +312,9 @@ function createClient(model: Model<"azure-openai-responses">, apiKey: string, op
|
|
|
312
312
|
maxRetries: resolveRetryBudget(options?.requestMaxRetries, 5),
|
|
313
313
|
defaultHeaders: headers,
|
|
314
314
|
baseURL: baseUrl,
|
|
315
|
-
fetch: onSseEvent
|
|
315
|
+
fetch: onSseEvent
|
|
316
|
+
? wrapFetchForSseDebug(baseFetch, event => onSseEvent(event, model, options?.attemptScope))
|
|
317
|
+
: baseFetch,
|
|
316
318
|
});
|
|
317
319
|
}
|
|
318
320
|
|
package/src/providers/cursor.ts
CHANGED
|
@@ -2632,7 +2632,7 @@ function buildGrpcRequest(
|
|
|
2632
2632
|
conversationId: state.conversationId,
|
|
2633
2633
|
});
|
|
2634
2634
|
|
|
2635
|
-
options?.onPayload?.(runRequest);
|
|
2635
|
+
options?.onPayload?.(runRequest, model, options?.attemptScope);
|
|
2636
2636
|
|
|
2637
2637
|
// Tools are sent later via requestContext (exec handshake)
|
|
2638
2638
|
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* DashScope Token Plan canonical request headers.
|
|
3
|
+
*
|
|
4
|
+
* Reproduces QwenLM/qwen-code's DashScopeOpenAICompatibleProvider.buildHeaders()
|
|
5
|
+
* defaultHeaders so the built-in `alibaba-token-plan` provider emits the same
|
|
6
|
+
* client identity / cache / auth-type fingerprint upstream sends. DashScope is
|
|
7
|
+
* compatibility-sensitive to this fingerprint; a non-identical set can cause
|
|
8
|
+
* request instability and affect first-event latency (gajae-code #3557).
|
|
9
|
+
*
|
|
10
|
+
* Upstream pin (reproduce EXACTLY here):
|
|
11
|
+
* Repository: QwenLM/qwen-code
|
|
12
|
+
* Commit: f4cd6e1d8bbb1c24e7e5d1a40187d8e28aa7c4fb
|
|
13
|
+
* Version: 0.21.1
|
|
14
|
+
* Source: packages/core/src/core/openaiContentGenerator/provider/dashscope.ts
|
|
15
|
+
* buildHeaders():
|
|
16
|
+
* const userAgent = `QwenCode/${version} (${process.platform}; ${process.arch})`;
|
|
17
|
+
* const defaultHeaders = {
|
|
18
|
+
* 'User-Agent': userAgent,
|
|
19
|
+
* 'X-DashScope-CacheControl': 'enable',
|
|
20
|
+
* 'X-DashScope-UserAgent': userAgent,
|
|
21
|
+
* 'X-DashScope-AuthType': authType,
|
|
22
|
+
* };
|
|
23
|
+
* return customHeaders ? { ...defaultHeaders, ...customHeaders } : defaultHeaders;
|
|
24
|
+
*
|
|
25
|
+
* The Token Plan preset authenticates with AuthType.USE_OPENAI ('openai'), so
|
|
26
|
+
* X-DashScope-AuthType is the constant 'openai'. Pin the version so an upstream
|
|
27
|
+
* bump is an explicit parity update rather than silent drift.
|
|
28
|
+
*/
|
|
29
|
+
export const QWEN_CODE_UPSTREAM_REPO = "QwenLM/qwen-code";
|
|
30
|
+
export const QWEN_CODE_UPSTREAM_COMMIT = "f4cd6e1d8bbb1c24e7e5d1a40187d8e28aa7c4fb";
|
|
31
|
+
export const QWEN_CODE_UPSTREAM_VERSION = "0.21.1";
|
|
32
|
+
|
|
33
|
+
// Upstream Token Plan preset uses AuthType.USE_OPENAI = 'openai'.
|
|
34
|
+
const QWEN_CODE_TOKEN_PLAN_AUTH_TYPE = "openai";
|
|
35
|
+
|
|
36
|
+
/**
|
|
37
|
+
* The Qwen Code CLI version string used in identity headers. Pinned to the
|
|
38
|
+
* upstream version at {@link QWEN_CODE_UPSTREAM_COMMIT}; change both together
|
|
39
|
+
* as an explicit parity update.
|
|
40
|
+
*/
|
|
41
|
+
export function qwenCodeUserAgent(version: string = QWEN_CODE_UPSTREAM_VERSION): string {
|
|
42
|
+
// process.platform / process.arch are read verbatim, matching upstream
|
|
43
|
+
// (e.g. "linux", "darwin", "win32"; "x64", "arm64"). No normalization.
|
|
44
|
+
return `QwenCode/${version} (${process.platform}; ${process.arch})`;
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
/**
|
|
48
|
+
* Canonical DashScope Token Plan headers (upstream defaultHeaders, no caller
|
|
49
|
+
* overrides applied). Exposed for tests/fixtures so the pinned wire set lives
|
|
50
|
+
* in exactly one place.
|
|
51
|
+
*/
|
|
52
|
+
export function dashscopeTokenPlanDefaultHeaders(
|
|
53
|
+
version: string = QWEN_CODE_UPSTREAM_VERSION,
|
|
54
|
+
): Readonly<Record<string, string>> {
|
|
55
|
+
const userAgent = qwenCodeUserAgent(version);
|
|
56
|
+
return Object.freeze({
|
|
57
|
+
"User-Agent": userAgent,
|
|
58
|
+
"X-DashScope-CacheControl": "enable",
|
|
59
|
+
"X-DashScope-UserAgent": userAgent,
|
|
60
|
+
"X-DashScope-AuthType": QWEN_CODE_TOKEN_PLAN_AUTH_TYPE,
|
|
61
|
+
});
|
|
62
|
+
}
|
|
63
|
+
|
|
64
|
+
/**
|
|
65
|
+
* Merge canonical DashScope Token Plan identity headers onto a caller's header
|
|
66
|
+
* map, reproducing upstream buildHeaders() precedence EXACTLY:
|
|
67
|
+
* `{ ...defaultHeaders, ...customHeaders }` — caller wins per header.
|
|
68
|
+
*
|
|
69
|
+
* This mirrors GJC's existing kimi-code injection order
|
|
70
|
+
* (`headers = { ...getKimiCommonHeaders(), ...headers }`): canonical identity as
|
|
71
|
+
* the base, caller-supplied headers overriding individual keys. A caller that
|
|
72
|
+
* pins `User-Agent` takes that key; the other canonicals still apply.
|
|
73
|
+
*
|
|
74
|
+
* A null/undefined `callerHeaders` returns the canonical set alone (upstream
|
|
75
|
+
* `customHeaders ? {...} : defaultHeaders` shortcut).
|
|
76
|
+
*/
|
|
77
|
+
export function mergeDashScopeTokenPlanHeaders(
|
|
78
|
+
callerHeaders: Record<string, string> | undefined,
|
|
79
|
+
version: string = QWEN_CODE_UPSTREAM_VERSION,
|
|
80
|
+
): Record<string, string> {
|
|
81
|
+
const defaults = dashscopeTokenPlanDefaultHeaders(version);
|
|
82
|
+
if (!callerHeaders) return { ...defaults };
|
|
83
|
+
return { ...defaults, ...callerHeaders };
|
|
84
|
+
}
|
|
@@ -279,6 +279,7 @@ export function streamGitLabDuo(
|
|
|
279
279
|
sessionId: options.sessionId,
|
|
280
280
|
providerSessionState: options.providerSessionState,
|
|
281
281
|
onPayload: options.onPayload,
|
|
282
|
+
attemptScope: options?.attemptScope,
|
|
282
283
|
onResponse: options.onResponse,
|
|
283
284
|
onSseEvent: options.onSseEvent,
|
|
284
285
|
fetch: options.fetch,
|
|
@@ -316,6 +317,7 @@ export function streamGitLabDuo(
|
|
|
316
317
|
sessionId: options.sessionId,
|
|
317
318
|
providerSessionState: options.providerSessionState,
|
|
318
319
|
onPayload: options.onPayload,
|
|
320
|
+
attemptScope: options?.attemptScope,
|
|
319
321
|
onResponse: options.onResponse,
|
|
320
322
|
onSseEvent: options.onSseEvent,
|
|
321
323
|
fetch: options.fetch,
|
|
@@ -348,6 +350,7 @@ export function streamGitLabDuo(
|
|
|
348
350
|
sessionId: options.sessionId,
|
|
349
351
|
providerSessionState: options.providerSessionState,
|
|
350
352
|
onPayload: options.onPayload,
|
|
353
|
+
attemptScope: options?.attemptScope,
|
|
351
354
|
onResponse: options.onResponse,
|
|
352
355
|
onSseEvent: options.onSseEvent,
|
|
353
356
|
fetch: options.fetch,
|
|
@@ -350,7 +350,7 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = (
|
|
|
350
350
|
const endpoints = baseUrl ? [baseUrl] : isAntigravity ? ANTIGRAVITY_ENDPOINT_FALLBACKS : [DEFAULT_ENDPOINT];
|
|
351
351
|
|
|
352
352
|
let requestBody = buildRequest(model, context, projectId, options, isAntigravity);
|
|
353
|
-
const replacementPayload = await options?.onPayload?.(requestBody, model);
|
|
353
|
+
const replacementPayload = await options?.onPayload?.(requestBody, model, options?.attemptScope);
|
|
354
354
|
if (replacementPayload !== undefined) {
|
|
355
355
|
requestBody = replacementPayload as typeof requestBody;
|
|
356
356
|
}
|
|
@@ -483,7 +483,12 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = (
|
|
|
483
483
|
for await (const chunk of readSseJson<CloudCodeAssistResponseChunk>(
|
|
484
484
|
activeResponse.body!,
|
|
485
485
|
options?.signal,
|
|
486
|
-
event =>
|
|
486
|
+
event =>
|
|
487
|
+
options?.onSseEvent?.(
|
|
488
|
+
{ event: event.event, data: event.data, raw: [...event.raw] },
|
|
489
|
+
model,
|
|
490
|
+
options?.attemptScope,
|
|
491
|
+
),
|
|
487
492
|
)) {
|
|
488
493
|
const responseData = chunk.response;
|
|
489
494
|
if (!responseData) continue;
|
|
@@ -835,7 +835,7 @@ export function streamGoogleGenAI<T extends "google-generative-ai" | "google-ver
|
|
|
835
835
|
try {
|
|
836
836
|
const plan = await prepare();
|
|
837
837
|
let params = plan.params;
|
|
838
|
-
const replacement = await options?.onPayload?.(params, model);
|
|
838
|
+
const replacement = await options?.onPayload?.(params, model, options?.attemptScope);
|
|
839
839
|
if (replacement !== undefined) {
|
|
840
840
|
params = replacement as GenerateContentParameters;
|
|
841
841
|
}
|
|
@@ -915,7 +915,11 @@ export function streamGoogleGenAI<T extends "google-generative-ai" | "google-ver
|
|
|
915
915
|
}
|
|
916
916
|
|
|
917
917
|
const googleStream = readSseJson<GenerateContentResponse>(response.body, options?.signal, event =>
|
|
918
|
-
options?.onSseEvent?.(
|
|
918
|
+
options?.onSseEvent?.(
|
|
919
|
+
{ event: event.event, data: event.data, raw: [...event.raw] },
|
|
920
|
+
model,
|
|
921
|
+
options?.attemptScope,
|
|
922
|
+
),
|
|
919
923
|
);
|
|
920
924
|
|
|
921
925
|
stream.push({ type: "start", partial: output });
|
package/src/providers/mock.ts
CHANGED
package/src/providers/ollama.ts
CHANGED
|
@@ -406,7 +406,7 @@ export const streamOllama: StreamFunction<"ollama-chat"> = (
|
|
|
406
406
|
const baseUrl = normalizeBaseUrl(model.baseUrl);
|
|
407
407
|
let body = createChatBody(model, context, options);
|
|
408
408
|
const sentForcedToolChoice = body.tool_choice === "required";
|
|
409
|
-
const replacementPayload = await options.onPayload?.(body, model);
|
|
409
|
+
const replacementPayload = await options.onPayload?.(body, model, options?.attemptScope);
|
|
410
410
|
if (replacementPayload !== undefined) {
|
|
411
411
|
body = replacementPayload as typeof body;
|
|
412
412
|
}
|
|
@@ -86,6 +86,7 @@ export function streamOpenAIAnthropicShim(
|
|
|
86
86
|
headers: mergedHeaders,
|
|
87
87
|
sessionId: options?.sessionId,
|
|
88
88
|
onPayload: options?.onPayload,
|
|
89
|
+
attemptScope: options?.attemptScope,
|
|
89
90
|
onResponse: options?.onResponse,
|
|
90
91
|
onSseEvent: options?.onSseEvent,
|
|
91
92
|
fetch: options?.fetch,
|
|
@@ -117,6 +118,7 @@ export function streamOpenAIAnthropicShim(
|
|
|
117
118
|
headers: mergedHeaders,
|
|
118
119
|
sessionId: options?.sessionId,
|
|
119
120
|
onPayload: options?.onPayload,
|
|
121
|
+
attemptScope: options?.attemptScope,
|
|
120
122
|
onResponse: options?.onResponse,
|
|
121
123
|
onSseEvent: options?.onSseEvent,
|
|
122
124
|
fetch: options?.fetch,
|
|
@@ -43,6 +43,7 @@ import {
|
|
|
43
43
|
createOpenAIResponsesHistoryPayload,
|
|
44
44
|
getOpenAIResponsesHistoryItems,
|
|
45
45
|
getOpenAIResponsesHistoryPayload,
|
|
46
|
+
neutralizeReservedControlTokens,
|
|
46
47
|
neutralizeResponsesInputControlTokens,
|
|
47
48
|
normalizeSystemPrompts,
|
|
48
49
|
sanitizeOpenAIResponsesHistoryItemsForReplay,
|
|
@@ -646,7 +647,7 @@ async function buildCodexRequestContext(
|
|
|
646
647
|
const url = resolveCodexResponsesUrl(baseUrl);
|
|
647
648
|
const promptCacheKey = normalizeOpenAIResponsesPromptCacheKey(options?.sessionId);
|
|
648
649
|
const transformedBody = await buildTransformedCodexRequestBody(model, context, options);
|
|
649
|
-
options?.onPayload?.(transformedBody);
|
|
650
|
+
options?.onPayload?.(transformedBody, model, options?.attemptScope);
|
|
650
651
|
|
|
651
652
|
const requestHeaders = { ...(model.headers ?? {}), ...(options?.headers ?? {}) };
|
|
652
653
|
const rawRequestDump: RawHttpRequestDump = {
|
|
@@ -746,7 +747,12 @@ async function buildTransformedCodexRequestBody(
|
|
|
746
747
|
}
|
|
747
748
|
}
|
|
748
749
|
|
|
749
|
-
|
|
750
|
+
// Neutralize leaked Harmony control tokens in the system prompt too:
|
|
751
|
+
// `params.instructions` and the developer messages prepended inside
|
|
752
|
+
// `transformRequestBody` bypass the `input` sanitizer above, so a poisoned
|
|
753
|
+
// system prompt rejects every turn with
|
|
754
|
+
// `Request blocked (code=invalid_prompt)`.
|
|
755
|
+
const systemPrompts = normalizeSystemPrompts(context.systemPrompt).map(neutralizeReservedControlTokens);
|
|
750
756
|
if (systemPrompts.length > 0) {
|
|
751
757
|
params.instructions = systemPrompts[0];
|
|
752
758
|
}
|
|
@@ -875,7 +881,7 @@ async function openCodexSseTransport(
|
|
|
875
881
|
body,
|
|
876
882
|
state,
|
|
877
883
|
requestSetup.requestSignal,
|
|
878
|
-
event => options?.onSseEvent?.(event, model),
|
|
884
|
+
event => options?.onSseEvent?.(event, model, options?.attemptScope),
|
|
879
885
|
options?.fetch,
|
|
880
886
|
options,
|
|
881
887
|
),
|
|
@@ -68,6 +68,7 @@ import {
|
|
|
68
68
|
resolveToolChoice,
|
|
69
69
|
} from "../utils/tool-choice-capability";
|
|
70
70
|
import { COMPOSER_EDIT_DISCIPLINE_PROMPT, isComposerHarnessModel } from "./composer-discipline";
|
|
71
|
+
import { mergeDashScopeTokenPlanHeaders } from "./dashscope-token-plan-headers";
|
|
71
72
|
import {
|
|
72
73
|
buildCopilotDynamicHeaders,
|
|
73
74
|
hasCopilotVisionInput,
|
|
@@ -481,6 +482,7 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
|
|
|
481
482
|
options?.requestMaxRetries,
|
|
482
483
|
options?.sessionId,
|
|
483
484
|
options?.maxRetryDelayMs,
|
|
485
|
+
options?.attemptScope,
|
|
484
486
|
);
|
|
485
487
|
const premiumRequestsTotal = copilotPremiumRequests;
|
|
486
488
|
getCapturedErrorResponse = captureErrorResponse;
|
|
@@ -503,7 +505,7 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
|
|
|
503
505
|
effectiveToolStrictModeOverride,
|
|
504
506
|
);
|
|
505
507
|
appliedToolStrictMode = toolStrictMode;
|
|
506
|
-
options?.onPayload?.(params);
|
|
508
|
+
options?.onPayload?.(params, undefined, options?.attemptScope);
|
|
507
509
|
rawRequestDump = {
|
|
508
510
|
provider: model.provider,
|
|
509
511
|
api: output.api,
|
|
@@ -1022,6 +1024,7 @@ async function createClient(
|
|
|
1022
1024
|
requestMaxRetries?: number,
|
|
1023
1025
|
sessionId?: string,
|
|
1024
1026
|
maxRetryDelayMs?: number,
|
|
1027
|
+
attemptScope?: import("../types.js").AttemptScopeRef,
|
|
1025
1028
|
): Promise<{
|
|
1026
1029
|
client: OpenAI;
|
|
1027
1030
|
copilotPremiumRequests: number | undefined;
|
|
@@ -1071,6 +1074,14 @@ async function createClient(
|
|
|
1071
1074
|
if (model.provider === "kimi-code") {
|
|
1072
1075
|
headers = { ...getKimiCommonHeaders(), ...headers };
|
|
1073
1076
|
}
|
|
1077
|
+
if (model.provider === "alibaba-token-plan") {
|
|
1078
|
+
// Emit Qwen Code's canonical DashScope request fingerprint (User-Agent /
|
|
1079
|
+
// X-DashScope-CacheControl / X-DashScope-UserAgent / X-DashScope-AuthType)
|
|
1080
|
+
// so DashScope treats the caller identically to upstream QwenLM/qwen-code.
|
|
1081
|
+
// Canonical identity is the base; caller headers win per key (upstream
|
|
1082
|
+
// `{...default, ...customHeaders}`). #3557.
|
|
1083
|
+
headers = mergeDashScopeTokenPlanHeaders(headers);
|
|
1084
|
+
}
|
|
1074
1085
|
headers = applyOpenAIRequestTransformHeaders(headers, model.requestTransform, `Gajae-Code/${packageJson.version}`);
|
|
1075
1086
|
let copilotPremiumRequests: number | undefined;
|
|
1076
1087
|
|
|
@@ -1136,7 +1147,7 @@ async function createClient(
|
|
|
1136
1147
|
`Gajae-Code/${packageJson.version}`,
|
|
1137
1148
|
);
|
|
1138
1149
|
const debugFetch = onSseEvent
|
|
1139
|
-
? wrapFetchForSseDebug(transformedFetch, event => onSseEvent(event, model))
|
|
1150
|
+
? wrapFetchForSseDebug(transformedFetch, event => onSseEvent(event, model, attemptScope))
|
|
1140
1151
|
: transformedFetch;
|
|
1141
1152
|
// Bound HTTP request timeout to roughly the first-event watchdog window.
|
|
1142
1153
|
// The OpenAI SDK's default is 10 minutes per attempt × `maxRetries`, which
|
|
@@ -27,6 +27,7 @@ import {
|
|
|
27
27
|
getOpenAIResponsesHistoryItems,
|
|
28
28
|
getOpenAIResponsesHistoryPayload,
|
|
29
29
|
isInvalidPromptError,
|
|
30
|
+
neutralizeReservedControlTokens,
|
|
30
31
|
neutralizeResponsesInputControlTokens,
|
|
31
32
|
normalizeSystemPrompts,
|
|
32
33
|
resolveCacheRetention,
|
|
@@ -60,6 +61,7 @@ import {
|
|
|
60
61
|
resolveToolChoice,
|
|
61
62
|
} from "../utils/tool-choice-capability";
|
|
62
63
|
import { COMPOSER_EDIT_DISCIPLINE_PROMPT, isComposerHarnessModel } from "./composer-discipline";
|
|
64
|
+
import { mergeDashScopeTokenPlanHeaders } from "./dashscope-token-plan-headers";
|
|
63
65
|
import {
|
|
64
66
|
buildCopilotDynamicHeaders,
|
|
65
67
|
hasCopilotVisionInput,
|
|
@@ -282,12 +284,13 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
|
|
|
282
284
|
options?.authCredentialType,
|
|
283
285
|
options?.requestMaxRetries,
|
|
284
286
|
options?.maxRetryDelayMs,
|
|
287
|
+
options?.attemptScope,
|
|
285
288
|
);
|
|
286
289
|
const premiumRequestsTotal = copilotPremiumRequests;
|
|
287
290
|
const providerSessionState = getOpenAIResponsesProviderSessionState(model, options?.providerSessionState);
|
|
288
291
|
const { params } = buildParams(model, context, options, providerSessionState, baseUrl);
|
|
289
292
|
const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs();
|
|
290
|
-
options?.onPayload?.(params);
|
|
293
|
+
options?.onPayload?.(params, undefined, options?.attemptScope);
|
|
291
294
|
rawRequestDump = {
|
|
292
295
|
provider: model.provider,
|
|
293
296
|
api: output.api,
|
|
@@ -431,6 +434,7 @@ function createClient(
|
|
|
431
434
|
authCredentialType?: OpenAIResponsesOptions["authCredentialType"],
|
|
432
435
|
requestMaxRetries?: number,
|
|
433
436
|
maxRetryDelayMs?: number,
|
|
437
|
+
attemptScope?: import("../types.js").AttemptScopeRef,
|
|
434
438
|
): {
|
|
435
439
|
client: OpenAI;
|
|
436
440
|
copilotPremiumRequests: number | undefined;
|
|
@@ -446,8 +450,17 @@ function createClient(
|
|
|
446
450
|
}
|
|
447
451
|
const rawApiKey = apiKey;
|
|
448
452
|
|
|
453
|
+
const baseHeaders =
|
|
454
|
+
model.provider === "alibaba-token-plan"
|
|
455
|
+
? // Emit Qwen Code's canonical DashScope request fingerprint (User-Agent /
|
|
456
|
+
// X-DashScope-CacheControl / X-DashScope-UserAgent / X-DashScope-AuthType)
|
|
457
|
+
// so DashScope treats the caller identically to upstream QwenLM/qwen-code.
|
|
458
|
+
// Canonical identity is the base; caller headers win per key (upstream
|
|
459
|
+
// `{...default, ...customHeaders}`). #3557.
|
|
460
|
+
mergeDashScopeTokenPlanHeaders({ ...(model.headers ?? {}), ...(extraHeaders ?? {}) })
|
|
461
|
+
: { ...(model.headers ?? {}), ...(extraHeaders ?? {}) };
|
|
449
462
|
const headers = applyOpenAIRequestTransformHeaders(
|
|
450
|
-
|
|
463
|
+
baseHeaders,
|
|
451
464
|
model.requestTransform,
|
|
452
465
|
`Gajae-Code/${packageJson.version}`,
|
|
453
466
|
);
|
|
@@ -491,7 +504,7 @@ function createClient(
|
|
|
491
504
|
maxRetries: resolveRetryBudget(requestMaxRetries, 5),
|
|
492
505
|
defaultHeaders: headers,
|
|
493
506
|
fetch: onSseEvent
|
|
494
|
-
? wrapFetchForSseDebug(transformedFetch, event => onSseEvent(event, model))
|
|
507
|
+
? wrapFetchForSseDebug(transformedFetch, event => onSseEvent(event, model, attemptScope))
|
|
495
508
|
: transformedFetch,
|
|
496
509
|
}),
|
|
497
510
|
copilotPremiumRequests,
|
|
@@ -523,7 +536,13 @@ function buildParams(
|
|
|
523
536
|
);
|
|
524
537
|
const messages: ResponseInput = neutralizeResponsesInputControlTokens(conversationMessages);
|
|
525
538
|
|
|
526
|
-
|
|
539
|
+
// Neutralize leaked Harmony control tokens in the system prompt too: the
|
|
540
|
+
// `instructions` field and developer-role messages bypass the `input`
|
|
541
|
+
// request-boundary sanitizer above, and a poisoned system prompt (e.g.
|
|
542
|
+
// injected project context quoting `<|channel|>` markers) rejects EVERY
|
|
543
|
+
// turn with `Request blocked (code=invalid_prompt)` — unrepairable by the
|
|
544
|
+
// history circuit breaker.
|
|
545
|
+
const systemPrompts = normalizeSystemPrompts(context.systemPrompt).map(neutralizeReservedControlTokens);
|
|
527
546
|
if (isComposerHarnessModel(model.id)) {
|
|
528
547
|
systemPrompts.unshift(COMPOSER_EDIT_DISCIPLINE_PROMPT);
|
|
529
548
|
}
|
package/src/stream.ts
CHANGED
|
@@ -705,6 +705,7 @@ function mapOptionsForApi<TApi extends Api>(
|
|
|
705
705
|
onPayload: options?.onPayload,
|
|
706
706
|
onResponse: options?.onResponse,
|
|
707
707
|
onSseEvent: options?.onSseEvent,
|
|
708
|
+
attemptScope: options?.attemptScope,
|
|
708
709
|
execHandlers: options?.execHandlers,
|
|
709
710
|
[managedAttemptValidated]: hasValidatedManagedAttempt(options),
|
|
710
711
|
};
|
package/src/types.ts
CHANGED
|
@@ -385,19 +385,29 @@ export interface StreamOptions {
|
|
|
385
385
|
/**
|
|
386
386
|
* Optional callback for inspecting or replacing provider payloads before sending.
|
|
387
387
|
* Return undefined to keep the payload unchanged.
|
|
388
|
+
* The `scope` parameter carries the per-attempt identity for execution attribution.
|
|
388
389
|
*/
|
|
389
|
-
onPayload?: (
|
|
390
|
+
onPayload?: (
|
|
391
|
+
payload: unknown,
|
|
392
|
+
model?: Model<Api>,
|
|
393
|
+
scope?: AttemptScopeRef,
|
|
394
|
+
) => unknown | undefined | Promise<unknown | undefined>;
|
|
390
395
|
/**
|
|
391
396
|
* Optional callback for provider response metadata after headers are received.
|
|
397
|
+
* The `scope` parameter carries the per-attempt identity for execution attribution.
|
|
392
398
|
*/
|
|
393
|
-
onResponse?: (
|
|
399
|
+
onResponse?: (
|
|
400
|
+
response: ProviderResponseMetadata,
|
|
401
|
+
model?: Model<Api>,
|
|
402
|
+
scope?: AttemptScopeRef,
|
|
403
|
+
) => void | Promise<void>;
|
|
394
404
|
/**
|
|
395
405
|
* Optional callback for raw Server-Sent Events as they arrive from HTTP streaming providers.
|
|
396
406
|
*
|
|
397
407
|
* Diagnostic only: provider implementations must ignore callback failures and must not
|
|
398
408
|
* let observers alter stream contents.
|
|
399
409
|
*/
|
|
400
|
-
onSseEvent?: (event: RawSseEvent, model?: Model<Api
|
|
410
|
+
onSseEvent?: (event: RawSseEvent, model?: Model<Api>, scope?: AttemptScopeRef) => void;
|
|
401
411
|
/**
|
|
402
412
|
* Optional override for the first streamed event watchdog in milliseconds.
|
|
403
413
|
* Set to 0 to disable the first-event watchdog for this request.
|
|
@@ -427,6 +437,23 @@ export interface StreamOptions {
|
|
|
427
437
|
authCredentialType?: "api_key" | "oauth";
|
|
428
438
|
/** Cursor exec/MCP tool handlers (cursor-agent only). */
|
|
429
439
|
execHandlers?: CursorExecHandlers;
|
|
440
|
+
/** Per-attempt identity for execution attribution. Threaded into onPayload/onResponse calls. */
|
|
441
|
+
attemptScope?: AttemptScopeRef;
|
|
442
|
+
}
|
|
443
|
+
|
|
444
|
+
/**
|
|
445
|
+
* Low-level structural carrier for per-attempt identity attribution.
|
|
446
|
+
*
|
|
447
|
+
* Defined in `packages/ai` so that {@link SimpleStreamOptions} and provider
|
|
448
|
+
* hook signatures can carry an attempt identity without a reverse dependency
|
|
449
|
+
* on `packages/agent`. The concrete `AttemptScope` in `packages/agent` is
|
|
450
|
+
* structurally assignable to this interface (same `attemptId` + `generation`
|
|
451
|
+
* + `lineage` fields).
|
|
452
|
+
*/
|
|
453
|
+
export interface AttemptScopeRef {
|
|
454
|
+
readonly attemptId: string;
|
|
455
|
+
readonly generation: number;
|
|
456
|
+
readonly lineage: string;
|
|
430
457
|
}
|
|
431
458
|
|
|
432
459
|
// Unified options with reasoning passed to streamSimple() and completeSimple()
|
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import type { Api, Model, ProviderResponseMetadata, StreamOptions } from "../types";
|
|
1
|
+
import type { Api, AttemptScopeRef, Model, ProviderResponseMetadata, StreamOptions } from "../types";
|
|
2
2
|
|
|
3
3
|
export function normalizeProviderResponse(
|
|
4
4
|
response: Response,
|
|
@@ -19,12 +19,12 @@ export function normalizeProviderResponse(
|
|
|
19
19
|
}
|
|
20
20
|
|
|
21
21
|
export async function notifyProviderResponse(
|
|
22
|
-
options:
|
|
22
|
+
options: { onResponse?: StreamOptions["onResponse"]; attemptScope?: AttemptScopeRef } | undefined,
|
|
23
23
|
response: Response,
|
|
24
24
|
model?: Model<Api>,
|
|
25
25
|
requestId?: string | null,
|
|
26
26
|
metadata?: Record<string, unknown>,
|
|
27
27
|
): Promise<void> {
|
|
28
28
|
if (!options?.onResponse) return;
|
|
29
|
-
await options.onResponse(normalizeProviderResponse(response, requestId, metadata), model);
|
|
29
|
+
await options.onResponse(normalizeProviderResponse(response, requestId, metadata), model, options.attemptScope);
|
|
30
30
|
}
|