@gajae-code/ai 0.12.4 → 0.12.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -2,6 +2,18 @@
2
2
 
3
3
  ## [Unreleased]
4
4
 
5
+ ## [0.12.6] - 2026-07-31
6
+
7
+ ## [0.12.5] - 2026-07-30
8
+ ### Fixed
9
+
10
+ - Alibaba Token Plan requests now carry Qwen Code's canonical DashScope request fingerprint on both transports. The built-in `alibaba-token-plan` provider (openai-responses `qwen3.8-max-preview` and openai-completions `glm-5.2`/`deepseek-v4-pro`) now emits the four upstream identity/cache/auth headers (`User-Agent`, `X-DashScope-CacheControl: enable`, `X-DashScope-UserAgent`, `X-DashScope-AuthType: openai`) matching `QwenLM/qwen-code` v0.21.1 (commit `f4cd6e1`) exactly, via a shared helper. DashScope is compatibility-sensitive to this client fingerprint, so a non-identical set can cause request instability and affect first-event latency. Caller headers still win per key (upstream `{...default, ...customHeaders}` precedence); non-Alibaba providers are byte-unchanged (#3557).
11
+
12
+ ### Added
13
+
14
+ - Reproducible Alibaba Token Plan header-parity A/B latency benchmark (`packages/ai/scripts/alibaba-token-plan-latency-ab.ts`): a fixed-seed interleaved A/B comparison of legacy vs Qwen-identical headers against a deterministic local HTTP server, reporting n/success/error/timeout and TTFT/total latency median/p90/p95/mean/stddev. No live credentials are required; a public-safe blocked-live-data receipt is included (`packages/ai/test/fixtures/alibaba-token-plan-latency-blocked-receipt.md`) (#3557).
15
+
16
+
5
17
  ## [0.12.4] - 2026-07-30
6
18
 
7
19
  ### Fixed
@@ -0,0 +1,57 @@
1
+ /**
2
+ * DashScope Token Plan canonical request headers.
3
+ *
4
+ * Reproduces QwenLM/qwen-code's DashScopeOpenAICompatibleProvider.buildHeaders()
5
+ * defaultHeaders so the built-in `alibaba-token-plan` provider emits the same
6
+ * client identity / cache / auth-type fingerprint upstream sends. DashScope is
7
+ * compatibility-sensitive to this fingerprint; a non-identical set can cause
8
+ * request instability and affect first-event latency (gajae-code #3557).
9
+ *
10
+ * Upstream pin (reproduce EXACTLY here):
11
+ * Repository: QwenLM/qwen-code
12
+ * Commit: f4cd6e1d8bbb1c24e7e5d1a40187d8e28aa7c4fb
13
+ * Version: 0.21.1
14
+ * Source: packages/core/src/core/openaiContentGenerator/provider/dashscope.ts
15
+ * buildHeaders():
16
+ * const userAgent = `QwenCode/${version} (${process.platform}; ${process.arch})`;
17
+ * const defaultHeaders = {
18
+ * 'User-Agent': userAgent,
19
+ * 'X-DashScope-CacheControl': 'enable',
20
+ * 'X-DashScope-UserAgent': userAgent,
21
+ * 'X-DashScope-AuthType': authType,
22
+ * };
23
+ * return customHeaders ? { ...defaultHeaders, ...customHeaders } : defaultHeaders;
24
+ *
25
+ * The Token Plan preset authenticates with AuthType.USE_OPENAI ('openai'), so
26
+ * X-DashScope-AuthType is the constant 'openai'. Pin the version so an upstream
27
+ * bump is an explicit parity update rather than silent drift.
28
+ */
29
+ export declare const QWEN_CODE_UPSTREAM_REPO = "QwenLM/qwen-code";
30
+ export declare const QWEN_CODE_UPSTREAM_COMMIT = "f4cd6e1d8bbb1c24e7e5d1a40187d8e28aa7c4fb";
31
+ export declare const QWEN_CODE_UPSTREAM_VERSION = "0.21.1";
32
+ /**
33
+ * The Qwen Code CLI version string used in identity headers. Pinned to the
34
+ * upstream version at {@link QWEN_CODE_UPSTREAM_COMMIT}; change both together
35
+ * as an explicit parity update.
36
+ */
37
+ export declare function qwenCodeUserAgent(version?: string): string;
38
+ /**
39
+ * Canonical DashScope Token Plan headers (upstream defaultHeaders, no caller
40
+ * overrides applied). Exposed for tests/fixtures so the pinned wire set lives
41
+ * in exactly one place.
42
+ */
43
+ export declare function dashscopeTokenPlanDefaultHeaders(version?: string): Readonly<Record<string, string>>;
44
+ /**
45
+ * Merge canonical DashScope Token Plan identity headers onto a caller's header
46
+ * map, reproducing upstream buildHeaders() precedence EXACTLY:
47
+ * `{ ...defaultHeaders, ...customHeaders }` — caller wins per header.
48
+ *
49
+ * This mirrors GJC's existing kimi-code injection order
50
+ * (`headers = { ...getKimiCommonHeaders(), ...headers }`): canonical identity as
51
+ * the base, caller-supplied headers overriding individual keys. A caller that
52
+ * pins `User-Agent` takes that key; the other canonicals still apply.
53
+ *
54
+ * A null/undefined `callerHeaders` returns the canonical set alone (upstream
55
+ * `customHeaders ? {...} : defaultHeaders` shortcut).
56
+ */
57
+ export declare function mergeDashScopeTokenPlanHeaders(callerHeaders: Record<string, string> | undefined, version?: string): Record<string, string>;
@@ -224,19 +224,21 @@ export interface StreamOptions {
224
224
  /**
225
225
  * Optional callback for inspecting or replacing provider payloads before sending.
226
226
  * Return undefined to keep the payload unchanged.
227
+ * The `scope` parameter carries the per-attempt identity for execution attribution.
227
228
  */
228
- onPayload?: (payload: unknown, model?: Model<Api>) => unknown | undefined | Promise<unknown | undefined>;
229
+ onPayload?: (payload: unknown, model?: Model<Api>, scope?: AttemptScopeRef) => unknown | undefined | Promise<unknown | undefined>;
229
230
  /**
230
231
  * Optional callback for provider response metadata after headers are received.
232
+ * The `scope` parameter carries the per-attempt identity for execution attribution.
231
233
  */
232
- onResponse?: (response: ProviderResponseMetadata, model?: Model<Api>) => void | Promise<void>;
234
+ onResponse?: (response: ProviderResponseMetadata, model?: Model<Api>, scope?: AttemptScopeRef) => void | Promise<void>;
233
235
  /**
234
236
  * Optional callback for raw Server-Sent Events as they arrive from HTTP streaming providers.
235
237
  *
236
238
  * Diagnostic only: provider implementations must ignore callback failures and must not
237
239
  * let observers alter stream contents.
238
240
  */
239
- onSseEvent?: (event: RawSseEvent, model?: Model<Api>) => void;
241
+ onSseEvent?: (event: RawSseEvent, model?: Model<Api>, scope?: AttemptScopeRef) => void;
240
242
  /**
241
243
  * Optional override for the first streamed event watchdog in milliseconds.
242
244
  * Set to 0 to disable the first-event watchdog for this request.
@@ -266,6 +268,22 @@ export interface StreamOptions {
266
268
  authCredentialType?: "api_key" | "oauth";
267
269
  /** Cursor exec/MCP tool handlers (cursor-agent only). */
268
270
  execHandlers?: CursorExecHandlers;
271
+ /** Per-attempt identity for execution attribution. Threaded into onPayload/onResponse calls. */
272
+ attemptScope?: AttemptScopeRef;
273
+ }
274
+ /**
275
+ * Low-level structural carrier for per-attempt identity attribution.
276
+ *
277
+ * Defined in `packages/ai` so that {@link SimpleStreamOptions} and provider
278
+ * hook signatures can carry an attempt identity without a reverse dependency
279
+ * on `packages/agent`. The concrete `AttemptScope` in `packages/agent` is
280
+ * structurally assignable to this interface (same `attemptId` + `generation`
281
+ * + `lineage` fields).
282
+ */
283
+ export interface AttemptScopeRef {
284
+ readonly attemptId: string;
285
+ readonly generation: number;
286
+ readonly lineage: string;
269
287
  }
270
288
  export interface SimpleStreamOptions extends StreamOptions {
271
289
  reasoning?: Effort;
@@ -1,3 +1,6 @@
1
- import type { Api, Model, ProviderResponseMetadata, StreamOptions } from "../types";
1
+ import type { Api, AttemptScopeRef, Model, ProviderResponseMetadata, StreamOptions } from "../types";
2
2
  export declare function normalizeProviderResponse(response: Response, requestId?: string | null, metadata?: Record<string, unknown>): ProviderResponseMetadata;
3
- export declare function notifyProviderResponse(options: Pick<StreamOptions, "onResponse"> | undefined, response: Response, model?: Model<Api>, requestId?: string | null, metadata?: Record<string, unknown>): Promise<void>;
3
+ export declare function notifyProviderResponse(options: {
4
+ onResponse?: StreamOptions["onResponse"];
5
+ attemptScope?: AttemptScopeRef;
6
+ } | undefined, response: Response, model?: Model<Api>, requestId?: string | null, metadata?: Record<string, unknown>): Promise<void>;
package/package.json CHANGED
@@ -1,7 +1,7 @@
1
1
  {
2
2
  "type": "module",
3
3
  "name": "@gajae-code/ai",
4
- "version": "0.12.4",
4
+ "version": "0.12.6",
5
5
  "description": "Unified LLM API with automatic model discovery and provider configuration",
6
6
  "homepage": "https://gajae-code.com",
7
7
  "author": "Yeachan-Heo and Gajae Code Contributors",
@@ -40,7 +40,7 @@
40
40
  "dependencies": {
41
41
  "@anthropic-ai/sdk": "^0.94.0",
42
42
  "@bufbuild/protobuf": "^2.12.0",
43
- "@gajae-code/utils": "0.12.4",
43
+ "@gajae-code/utils": "0.12.6",
44
44
  "openai": "^6.36.0",
45
45
  "partial-json": "^0.1.7",
46
46
  "zod": "4.4.3"
@@ -228,7 +228,7 @@ export const streamBedrock: StreamFunction<"bedrock-converse-stream"> = (
228
228
  toolConfig,
229
229
  additionalModelRequestFields,
230
230
  };
231
- options?.onPayload?.(commandInput);
231
+ options?.onPayload?.(commandInput, model, options?.attemptScope);
232
232
 
233
233
  const host = `bedrock-runtime.${region}.amazonaws.com`;
234
234
  const url = `https://${host}/model/${encodeURIComponent(model.id)}/converse-stream`;
@@ -1348,7 +1348,9 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1348
1348
  dynamicHeaders: copilotDynamicHeaders?.headers,
1349
1349
  isOAuth: options?.isOAuth,
1350
1350
  hasTools: !!context.tools?.length,
1351
- onSseEvent: options?.onSseEvent,
1351
+ onSseEvent: options?.onSseEvent
1352
+ ? event => options.onSseEvent!(event, model, options?.attemptScope)
1353
+ : undefined,
1352
1354
  fetch: options?.fetch,
1353
1355
  requestMaxRetries: options?.requestMaxRetries,
1354
1356
  maxRetryDelayMs: options?.maxRetryDelayMs,
@@ -1386,7 +1388,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1386
1388
  if (dropFastMode) {
1387
1389
  dropAnthropicFastMode(nextParams);
1388
1390
  }
1389
- const replacementPayload = await options?.onPayload?.(nextParams, model);
1391
+ const replacementPayload = await options?.onPayload?.(nextParams, model, options?.attemptScope);
1390
1392
  if (replacementPayload !== undefined) {
1391
1393
  nextParams = replacementPayload as typeof nextParams;
1392
1394
  }
@@ -1489,7 +1491,7 @@ export const streamAnthropic: StreamFunction<"anthropic-messages"> = (
1489
1491
  } = await getAnthropicStreamResponse(
1490
1492
  anthropicRequest,
1491
1493
  requestSignal,
1492
- options?.client ? event => options?.onSseEvent?.(event, model) : undefined,
1494
+ options?.client ? event => options?.onSseEvent?.(event, model, options?.attemptScope) : undefined,
1493
1495
  );
1494
1496
  await notifyProviderResponse(options, response, model, requestId);
1495
1497
  let sawEvent = false;
@@ -126,7 +126,7 @@ export const streamAzureOpenAIResponses: StreamFunction<"azure-openai-responses"
126
126
  const { baseUrl } = resolveAzureConfig(model, options);
127
127
  const params = buildParams(model, context, options, deploymentName, baseUrl);
128
128
  const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs();
129
- options?.onPayload?.(params);
129
+ options?.onPayload?.(params, model, options?.attemptScope);
130
130
  rawRequestDump = {
131
131
  provider: model.provider,
132
132
  api: output.api,
@@ -312,7 +312,9 @@ function createClient(model: Model<"azure-openai-responses">, apiKey: string, op
312
312
  maxRetries: resolveRetryBudget(options?.requestMaxRetries, 5),
313
313
  defaultHeaders: headers,
314
314
  baseURL: baseUrl,
315
- fetch: onSseEvent ? wrapFetchForSseDebug(baseFetch, event => onSseEvent(event, model)) : baseFetch,
315
+ fetch: onSseEvent
316
+ ? wrapFetchForSseDebug(baseFetch, event => onSseEvent(event, model, options?.attemptScope))
317
+ : baseFetch,
316
318
  });
317
319
  }
318
320
 
@@ -2632,7 +2632,7 @@ function buildGrpcRequest(
2632
2632
  conversationId: state.conversationId,
2633
2633
  });
2634
2634
 
2635
- options?.onPayload?.(runRequest);
2635
+ options?.onPayload?.(runRequest, model, options?.attemptScope);
2636
2636
 
2637
2637
  // Tools are sent later via requestContext (exec handshake)
2638
2638
 
@@ -0,0 +1,84 @@
1
+ /**
2
+ * DashScope Token Plan canonical request headers.
3
+ *
4
+ * Reproduces QwenLM/qwen-code's DashScopeOpenAICompatibleProvider.buildHeaders()
5
+ * defaultHeaders so the built-in `alibaba-token-plan` provider emits the same
6
+ * client identity / cache / auth-type fingerprint upstream sends. DashScope is
7
+ * compatibility-sensitive to this fingerprint; a non-identical set can cause
8
+ * request instability and affect first-event latency (gajae-code #3557).
9
+ *
10
+ * Upstream pin (reproduce EXACTLY here):
11
+ * Repository: QwenLM/qwen-code
12
+ * Commit: f4cd6e1d8bbb1c24e7e5d1a40187d8e28aa7c4fb
13
+ * Version: 0.21.1
14
+ * Source: packages/core/src/core/openaiContentGenerator/provider/dashscope.ts
15
+ * buildHeaders():
16
+ * const userAgent = `QwenCode/${version} (${process.platform}; ${process.arch})`;
17
+ * const defaultHeaders = {
18
+ * 'User-Agent': userAgent,
19
+ * 'X-DashScope-CacheControl': 'enable',
20
+ * 'X-DashScope-UserAgent': userAgent,
21
+ * 'X-DashScope-AuthType': authType,
22
+ * };
23
+ * return customHeaders ? { ...defaultHeaders, ...customHeaders } : defaultHeaders;
24
+ *
25
+ * The Token Plan preset authenticates with AuthType.USE_OPENAI ('openai'), so
26
+ * X-DashScope-AuthType is the constant 'openai'. Pin the version so an upstream
27
+ * bump is an explicit parity update rather than silent drift.
28
+ */
29
+ export const QWEN_CODE_UPSTREAM_REPO = "QwenLM/qwen-code";
30
+ export const QWEN_CODE_UPSTREAM_COMMIT = "f4cd6e1d8bbb1c24e7e5d1a40187d8e28aa7c4fb";
31
+ export const QWEN_CODE_UPSTREAM_VERSION = "0.21.1";
32
+
33
+ // Upstream Token Plan preset uses AuthType.USE_OPENAI = 'openai'.
34
+ const QWEN_CODE_TOKEN_PLAN_AUTH_TYPE = "openai";
35
+
36
+ /**
37
+ * The Qwen Code CLI version string used in identity headers. Pinned to the
38
+ * upstream version at {@link QWEN_CODE_UPSTREAM_COMMIT}; change both together
39
+ * as an explicit parity update.
40
+ */
41
+ export function qwenCodeUserAgent(version: string = QWEN_CODE_UPSTREAM_VERSION): string {
42
+ // process.platform / process.arch are read verbatim, matching upstream
43
+ // (e.g. "linux", "darwin", "win32"; "x64", "arm64"). No normalization.
44
+ return `QwenCode/${version} (${process.platform}; ${process.arch})`;
45
+ }
46
+
47
+ /**
48
+ * Canonical DashScope Token Plan headers (upstream defaultHeaders, no caller
49
+ * overrides applied). Exposed for tests/fixtures so the pinned wire set lives
50
+ * in exactly one place.
51
+ */
52
+ export function dashscopeTokenPlanDefaultHeaders(
53
+ version: string = QWEN_CODE_UPSTREAM_VERSION,
54
+ ): Readonly<Record<string, string>> {
55
+ const userAgent = qwenCodeUserAgent(version);
56
+ return Object.freeze({
57
+ "User-Agent": userAgent,
58
+ "X-DashScope-CacheControl": "enable",
59
+ "X-DashScope-UserAgent": userAgent,
60
+ "X-DashScope-AuthType": QWEN_CODE_TOKEN_PLAN_AUTH_TYPE,
61
+ });
62
+ }
63
+
64
+ /**
65
+ * Merge canonical DashScope Token Plan identity headers onto a caller's header
66
+ * map, reproducing upstream buildHeaders() precedence EXACTLY:
67
+ * `{ ...defaultHeaders, ...customHeaders }` — caller wins per header.
68
+ *
69
+ * This mirrors GJC's existing kimi-code injection order
70
+ * (`headers = { ...getKimiCommonHeaders(), ...headers }`): canonical identity as
71
+ * the base, caller-supplied headers overriding individual keys. A caller that
72
+ * pins `User-Agent` takes that key; the other canonicals still apply.
73
+ *
74
+ * A null/undefined `callerHeaders` returns the canonical set alone (upstream
75
+ * `customHeaders ? {...} : defaultHeaders` shortcut).
76
+ */
77
+ export function mergeDashScopeTokenPlanHeaders(
78
+ callerHeaders: Record<string, string> | undefined,
79
+ version: string = QWEN_CODE_UPSTREAM_VERSION,
80
+ ): Record<string, string> {
81
+ const defaults = dashscopeTokenPlanDefaultHeaders(version);
82
+ if (!callerHeaders) return { ...defaults };
83
+ return { ...defaults, ...callerHeaders };
84
+ }
@@ -279,6 +279,7 @@ export function streamGitLabDuo(
279
279
  sessionId: options.sessionId,
280
280
  providerSessionState: options.providerSessionState,
281
281
  onPayload: options.onPayload,
282
+ attemptScope: options?.attemptScope,
282
283
  onResponse: options.onResponse,
283
284
  onSseEvent: options.onSseEvent,
284
285
  fetch: options.fetch,
@@ -316,6 +317,7 @@ export function streamGitLabDuo(
316
317
  sessionId: options.sessionId,
317
318
  providerSessionState: options.providerSessionState,
318
319
  onPayload: options.onPayload,
320
+ attemptScope: options?.attemptScope,
319
321
  onResponse: options.onResponse,
320
322
  onSseEvent: options.onSseEvent,
321
323
  fetch: options.fetch,
@@ -348,6 +350,7 @@ export function streamGitLabDuo(
348
350
  sessionId: options.sessionId,
349
351
  providerSessionState: options.providerSessionState,
350
352
  onPayload: options.onPayload,
353
+ attemptScope: options?.attemptScope,
351
354
  onResponse: options.onResponse,
352
355
  onSseEvent: options.onSseEvent,
353
356
  fetch: options.fetch,
@@ -350,7 +350,7 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = (
350
350
  const endpoints = baseUrl ? [baseUrl] : isAntigravity ? ANTIGRAVITY_ENDPOINT_FALLBACKS : [DEFAULT_ENDPOINT];
351
351
 
352
352
  let requestBody = buildRequest(model, context, projectId, options, isAntigravity);
353
- const replacementPayload = await options?.onPayload?.(requestBody, model);
353
+ const replacementPayload = await options?.onPayload?.(requestBody, model, options?.attemptScope);
354
354
  if (replacementPayload !== undefined) {
355
355
  requestBody = replacementPayload as typeof requestBody;
356
356
  }
@@ -483,7 +483,12 @@ export const streamGoogleGeminiCli: StreamFunction<"google-gemini-cli"> = (
483
483
  for await (const chunk of readSseJson<CloudCodeAssistResponseChunk>(
484
484
  activeResponse.body!,
485
485
  options?.signal,
486
- event => options?.onSseEvent?.({ event: event.event, data: event.data, raw: [...event.raw] }, model),
486
+ event =>
487
+ options?.onSseEvent?.(
488
+ { event: event.event, data: event.data, raw: [...event.raw] },
489
+ model,
490
+ options?.attemptScope,
491
+ ),
487
492
  )) {
488
493
  const responseData = chunk.response;
489
494
  if (!responseData) continue;
@@ -835,7 +835,7 @@ export function streamGoogleGenAI<T extends "google-generative-ai" | "google-ver
835
835
  try {
836
836
  const plan = await prepare();
837
837
  let params = plan.params;
838
- const replacement = await options?.onPayload?.(params, model);
838
+ const replacement = await options?.onPayload?.(params, model, options?.attemptScope);
839
839
  if (replacement !== undefined) {
840
840
  params = replacement as GenerateContentParameters;
841
841
  }
@@ -915,7 +915,11 @@ export function streamGoogleGenAI<T extends "google-generative-ai" | "google-ver
915
915
  }
916
916
 
917
917
  const googleStream = readSseJson<GenerateContentResponse>(response.body, options?.signal, event =>
918
- options?.onSseEvent?.({ event: event.event, data: event.data, raw: [...event.raw] }, model),
918
+ options?.onSseEvent?.(
919
+ { event: event.event, data: event.data, raw: [...event.raw] },
920
+ model,
921
+ options?.attemptScope,
922
+ ),
919
923
  );
920
924
 
921
925
  stream.push({ type: "start", partial: output });
@@ -327,6 +327,7 @@ async function runMock(
327
327
  ...(response.responseRequestId !== undefined ? { requestId: response.responseRequestId } : {}),
328
328
  },
329
329
  model,
330
+ options.attemptScope,
330
331
  );
331
332
  } catch (err) {
332
333
  stream.fail(err);
@@ -406,7 +406,7 @@ export const streamOllama: StreamFunction<"ollama-chat"> = (
406
406
  const baseUrl = normalizeBaseUrl(model.baseUrl);
407
407
  let body = createChatBody(model, context, options);
408
408
  const sentForcedToolChoice = body.tool_choice === "required";
409
- const replacementPayload = await options.onPayload?.(body, model);
409
+ const replacementPayload = await options.onPayload?.(body, model, options?.attemptScope);
410
410
  if (replacementPayload !== undefined) {
411
411
  body = replacementPayload as typeof body;
412
412
  }
@@ -86,6 +86,7 @@ export function streamOpenAIAnthropicShim(
86
86
  headers: mergedHeaders,
87
87
  sessionId: options?.sessionId,
88
88
  onPayload: options?.onPayload,
89
+ attemptScope: options?.attemptScope,
89
90
  onResponse: options?.onResponse,
90
91
  onSseEvent: options?.onSseEvent,
91
92
  fetch: options?.fetch,
@@ -117,6 +118,7 @@ export function streamOpenAIAnthropicShim(
117
118
  headers: mergedHeaders,
118
119
  sessionId: options?.sessionId,
119
120
  onPayload: options?.onPayload,
121
+ attemptScope: options?.attemptScope,
120
122
  onResponse: options?.onResponse,
121
123
  onSseEvent: options?.onSseEvent,
122
124
  fetch: options?.fetch,
@@ -43,6 +43,7 @@ import {
43
43
  createOpenAIResponsesHistoryPayload,
44
44
  getOpenAIResponsesHistoryItems,
45
45
  getOpenAIResponsesHistoryPayload,
46
+ neutralizeReservedControlTokens,
46
47
  neutralizeResponsesInputControlTokens,
47
48
  normalizeSystemPrompts,
48
49
  sanitizeOpenAIResponsesHistoryItemsForReplay,
@@ -646,7 +647,7 @@ async function buildCodexRequestContext(
646
647
  const url = resolveCodexResponsesUrl(baseUrl);
647
648
  const promptCacheKey = normalizeOpenAIResponsesPromptCacheKey(options?.sessionId);
648
649
  const transformedBody = await buildTransformedCodexRequestBody(model, context, options);
649
- options?.onPayload?.(transformedBody);
650
+ options?.onPayload?.(transformedBody, model, options?.attemptScope);
650
651
 
651
652
  const requestHeaders = { ...(model.headers ?? {}), ...(options?.headers ?? {}) };
652
653
  const rawRequestDump: RawHttpRequestDump = {
@@ -746,7 +747,12 @@ async function buildTransformedCodexRequestBody(
746
747
  }
747
748
  }
748
749
 
749
- const systemPrompts = normalizeSystemPrompts(context.systemPrompt);
750
+ // Neutralize leaked Harmony control tokens in the system prompt too:
751
+ // `params.instructions` and the developer messages prepended inside
752
+ // `transformRequestBody` bypass the `input` sanitizer above, so a poisoned
753
+ // system prompt rejects every turn with
754
+ // `Request blocked (code=invalid_prompt)`.
755
+ const systemPrompts = normalizeSystemPrompts(context.systemPrompt).map(neutralizeReservedControlTokens);
750
756
  if (systemPrompts.length > 0) {
751
757
  params.instructions = systemPrompts[0];
752
758
  }
@@ -875,7 +881,7 @@ async function openCodexSseTransport(
875
881
  body,
876
882
  state,
877
883
  requestSetup.requestSignal,
878
- event => options?.onSseEvent?.(event, model),
884
+ event => options?.onSseEvent?.(event, model, options?.attemptScope),
879
885
  options?.fetch,
880
886
  options,
881
887
  ),
@@ -68,6 +68,7 @@ import {
68
68
  resolveToolChoice,
69
69
  } from "../utils/tool-choice-capability";
70
70
  import { COMPOSER_EDIT_DISCIPLINE_PROMPT, isComposerHarnessModel } from "./composer-discipline";
71
+ import { mergeDashScopeTokenPlanHeaders } from "./dashscope-token-plan-headers";
71
72
  import {
72
73
  buildCopilotDynamicHeaders,
73
74
  hasCopilotVisionInput,
@@ -481,6 +482,7 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
481
482
  options?.requestMaxRetries,
482
483
  options?.sessionId,
483
484
  options?.maxRetryDelayMs,
485
+ options?.attemptScope,
484
486
  );
485
487
  const premiumRequestsTotal = copilotPremiumRequests;
486
488
  getCapturedErrorResponse = captureErrorResponse;
@@ -503,7 +505,7 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
503
505
  effectiveToolStrictModeOverride,
504
506
  );
505
507
  appliedToolStrictMode = toolStrictMode;
506
- options?.onPayload?.(params);
508
+ options?.onPayload?.(params, undefined, options?.attemptScope);
507
509
  rawRequestDump = {
508
510
  provider: model.provider,
509
511
  api: output.api,
@@ -1022,6 +1024,7 @@ async function createClient(
1022
1024
  requestMaxRetries?: number,
1023
1025
  sessionId?: string,
1024
1026
  maxRetryDelayMs?: number,
1027
+ attemptScope?: import("../types.js").AttemptScopeRef,
1025
1028
  ): Promise<{
1026
1029
  client: OpenAI;
1027
1030
  copilotPremiumRequests: number | undefined;
@@ -1071,6 +1074,14 @@ async function createClient(
1071
1074
  if (model.provider === "kimi-code") {
1072
1075
  headers = { ...getKimiCommonHeaders(), ...headers };
1073
1076
  }
1077
+ if (model.provider === "alibaba-token-plan") {
1078
+ // Emit Qwen Code's canonical DashScope request fingerprint (User-Agent /
1079
+ // X-DashScope-CacheControl / X-DashScope-UserAgent / X-DashScope-AuthType)
1080
+ // so DashScope treats the caller identically to upstream QwenLM/qwen-code.
1081
+ // Canonical identity is the base; caller headers win per key (upstream
1082
+ // `{...default, ...customHeaders}`). #3557.
1083
+ headers = mergeDashScopeTokenPlanHeaders(headers);
1084
+ }
1074
1085
  headers = applyOpenAIRequestTransformHeaders(headers, model.requestTransform, `Gajae-Code/${packageJson.version}`);
1075
1086
  let copilotPremiumRequests: number | undefined;
1076
1087
 
@@ -1136,7 +1147,7 @@ async function createClient(
1136
1147
  `Gajae-Code/${packageJson.version}`,
1137
1148
  );
1138
1149
  const debugFetch = onSseEvent
1139
- ? wrapFetchForSseDebug(transformedFetch, event => onSseEvent(event, model))
1150
+ ? wrapFetchForSseDebug(transformedFetch, event => onSseEvent(event, model, attemptScope))
1140
1151
  : transformedFetch;
1141
1152
  // Bound HTTP request timeout to roughly the first-event watchdog window.
1142
1153
  // The OpenAI SDK's default is 10 minutes per attempt × `maxRetries`, which
@@ -27,6 +27,7 @@ import {
27
27
  getOpenAIResponsesHistoryItems,
28
28
  getOpenAIResponsesHistoryPayload,
29
29
  isInvalidPromptError,
30
+ neutralizeReservedControlTokens,
30
31
  neutralizeResponsesInputControlTokens,
31
32
  normalizeSystemPrompts,
32
33
  resolveCacheRetention,
@@ -60,6 +61,7 @@ import {
60
61
  resolveToolChoice,
61
62
  } from "../utils/tool-choice-capability";
62
63
  import { COMPOSER_EDIT_DISCIPLINE_PROMPT, isComposerHarnessModel } from "./composer-discipline";
64
+ import { mergeDashScopeTokenPlanHeaders } from "./dashscope-token-plan-headers";
63
65
  import {
64
66
  buildCopilotDynamicHeaders,
65
67
  hasCopilotVisionInput,
@@ -282,12 +284,13 @@ export const streamOpenAIResponses: StreamFunction<"openai-responses"> = (
282
284
  options?.authCredentialType,
283
285
  options?.requestMaxRetries,
284
286
  options?.maxRetryDelayMs,
287
+ options?.attemptScope,
285
288
  );
286
289
  const premiumRequestsTotal = copilotPremiumRequests;
287
290
  const providerSessionState = getOpenAIResponsesProviderSessionState(model, options?.providerSessionState);
288
291
  const { params } = buildParams(model, context, options, providerSessionState, baseUrl);
289
292
  const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs();
290
- options?.onPayload?.(params);
293
+ options?.onPayload?.(params, undefined, options?.attemptScope);
291
294
  rawRequestDump = {
292
295
  provider: model.provider,
293
296
  api: output.api,
@@ -431,6 +434,7 @@ function createClient(
431
434
  authCredentialType?: OpenAIResponsesOptions["authCredentialType"],
432
435
  requestMaxRetries?: number,
433
436
  maxRetryDelayMs?: number,
437
+ attemptScope?: import("../types.js").AttemptScopeRef,
434
438
  ): {
435
439
  client: OpenAI;
436
440
  copilotPremiumRequests: number | undefined;
@@ -446,8 +450,17 @@ function createClient(
446
450
  }
447
451
  const rawApiKey = apiKey;
448
452
 
453
+ const baseHeaders =
454
+ model.provider === "alibaba-token-plan"
455
+ ? // Emit Qwen Code's canonical DashScope request fingerprint (User-Agent /
456
+ // X-DashScope-CacheControl / X-DashScope-UserAgent / X-DashScope-AuthType)
457
+ // so DashScope treats the caller identically to upstream QwenLM/qwen-code.
458
+ // Canonical identity is the base; caller headers win per key (upstream
459
+ // `{...default, ...customHeaders}`). #3557.
460
+ mergeDashScopeTokenPlanHeaders({ ...(model.headers ?? {}), ...(extraHeaders ?? {}) })
461
+ : { ...(model.headers ?? {}), ...(extraHeaders ?? {}) };
449
462
  const headers = applyOpenAIRequestTransformHeaders(
450
- { ...(model.headers ?? {}), ...(extraHeaders ?? {}) },
463
+ baseHeaders,
451
464
  model.requestTransform,
452
465
  `Gajae-Code/${packageJson.version}`,
453
466
  );
@@ -491,7 +504,7 @@ function createClient(
491
504
  maxRetries: resolveRetryBudget(requestMaxRetries, 5),
492
505
  defaultHeaders: headers,
493
506
  fetch: onSseEvent
494
- ? wrapFetchForSseDebug(transformedFetch, event => onSseEvent(event, model))
507
+ ? wrapFetchForSseDebug(transformedFetch, event => onSseEvent(event, model, attemptScope))
495
508
  : transformedFetch,
496
509
  }),
497
510
  copilotPremiumRequests,
@@ -523,7 +536,13 @@ function buildParams(
523
536
  );
524
537
  const messages: ResponseInput = neutralizeResponsesInputControlTokens(conversationMessages);
525
538
 
526
- const systemPrompts = normalizeSystemPrompts(context.systemPrompt);
539
+ // Neutralize leaked Harmony control tokens in the system prompt too: the
540
+ // `instructions` field and developer-role messages bypass the `input`
541
+ // request-boundary sanitizer above, and a poisoned system prompt (e.g.
542
+ // injected project context quoting `<|channel|>` markers) rejects EVERY
543
+ // turn with `Request blocked (code=invalid_prompt)` — unrepairable by the
544
+ // history circuit breaker.
545
+ const systemPrompts = normalizeSystemPrompts(context.systemPrompt).map(neutralizeReservedControlTokens);
527
546
  if (isComposerHarnessModel(model.id)) {
528
547
  systemPrompts.unshift(COMPOSER_EDIT_DISCIPLINE_PROMPT);
529
548
  }
package/src/stream.ts CHANGED
@@ -705,6 +705,7 @@ function mapOptionsForApi<TApi extends Api>(
705
705
  onPayload: options?.onPayload,
706
706
  onResponse: options?.onResponse,
707
707
  onSseEvent: options?.onSseEvent,
708
+ attemptScope: options?.attemptScope,
708
709
  execHandlers: options?.execHandlers,
709
710
  [managedAttemptValidated]: hasValidatedManagedAttempt(options),
710
711
  };
package/src/types.ts CHANGED
@@ -385,19 +385,29 @@ export interface StreamOptions {
385
385
  /**
386
386
  * Optional callback for inspecting or replacing provider payloads before sending.
387
387
  * Return undefined to keep the payload unchanged.
388
+ * The `scope` parameter carries the per-attempt identity for execution attribution.
388
389
  */
389
- onPayload?: (payload: unknown, model?: Model<Api>) => unknown | undefined | Promise<unknown | undefined>;
390
+ onPayload?: (
391
+ payload: unknown,
392
+ model?: Model<Api>,
393
+ scope?: AttemptScopeRef,
394
+ ) => unknown | undefined | Promise<unknown | undefined>;
390
395
  /**
391
396
  * Optional callback for provider response metadata after headers are received.
397
+ * The `scope` parameter carries the per-attempt identity for execution attribution.
392
398
  */
393
- onResponse?: (response: ProviderResponseMetadata, model?: Model<Api>) => void | Promise<void>;
399
+ onResponse?: (
400
+ response: ProviderResponseMetadata,
401
+ model?: Model<Api>,
402
+ scope?: AttemptScopeRef,
403
+ ) => void | Promise<void>;
394
404
  /**
395
405
  * Optional callback for raw Server-Sent Events as they arrive from HTTP streaming providers.
396
406
  *
397
407
  * Diagnostic only: provider implementations must ignore callback failures and must not
398
408
  * let observers alter stream contents.
399
409
  */
400
- onSseEvent?: (event: RawSseEvent, model?: Model<Api>) => void;
410
+ onSseEvent?: (event: RawSseEvent, model?: Model<Api>, scope?: AttemptScopeRef) => void;
401
411
  /**
402
412
  * Optional override for the first streamed event watchdog in milliseconds.
403
413
  * Set to 0 to disable the first-event watchdog for this request.
@@ -427,6 +437,23 @@ export interface StreamOptions {
427
437
  authCredentialType?: "api_key" | "oauth";
428
438
  /** Cursor exec/MCP tool handlers (cursor-agent only). */
429
439
  execHandlers?: CursorExecHandlers;
440
+ /** Per-attempt identity for execution attribution. Threaded into onPayload/onResponse calls. */
441
+ attemptScope?: AttemptScopeRef;
442
+ }
443
+
444
+ /**
445
+ * Low-level structural carrier for per-attempt identity attribution.
446
+ *
447
+ * Defined in `packages/ai` so that {@link SimpleStreamOptions} and provider
448
+ * hook signatures can carry an attempt identity without a reverse dependency
449
+ * on `packages/agent`. The concrete `AttemptScope` in `packages/agent` is
450
+ * structurally assignable to this interface (same `attemptId` + `generation`
451
+ * + `lineage` fields).
452
+ */
453
+ export interface AttemptScopeRef {
454
+ readonly attemptId: string;
455
+ readonly generation: number;
456
+ readonly lineage: string;
430
457
  }
431
458
 
432
459
  // Unified options with reasoning passed to streamSimple() and completeSimple()
@@ -1,4 +1,4 @@
1
- import type { Api, Model, ProviderResponseMetadata, StreamOptions } from "../types";
1
+ import type { Api, AttemptScopeRef, Model, ProviderResponseMetadata, StreamOptions } from "../types";
2
2
 
3
3
  export function normalizeProviderResponse(
4
4
  response: Response,
@@ -19,12 +19,12 @@ export function normalizeProviderResponse(
19
19
  }
20
20
 
21
21
  export async function notifyProviderResponse(
22
- options: Pick<StreamOptions, "onResponse"> | undefined,
22
+ options: { onResponse?: StreamOptions["onResponse"]; attemptScope?: AttemptScopeRef } | undefined,
23
23
  response: Response,
24
24
  model?: Model<Api>,
25
25
  requestId?: string | null,
26
26
  metadata?: Record<string, unknown>,
27
27
  ): Promise<void> {
28
28
  if (!options?.onResponse) return;
29
- await options.onResponse(normalizeProviderResponse(response, requestId, metadata), model);
29
+ await options.onResponse(normalizeProviderResponse(response, requestId, metadata), model, options.attemptScope);
30
30
  }