@gajae-code/ai 0.17.1 → 0.17.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +117 -0
- package/dist/types/auth-gateway/server.d.ts +23 -1
- package/dist/types/auth-storage.d.ts +12 -1
- package/dist/types/model-thinking.d.ts +10 -6
- package/dist/types/provider-models/openai-compat.d.ts +2 -2
- package/dist/types/providers/anthropic.d.ts +1 -1
- package/dist/types/providers/cursor.d.ts +10 -0
- package/dist/types/providers/openai-completions.d.ts +9 -1
- package/dist/types/types.d.ts +16 -0
- package/dist/types/utils/discovery/openai-compatible.d.ts +10 -0
- package/dist/types/utils/fallback-transport.d.ts +4 -0
- package/dist/types/utils/h2-fetch.d.ts +8 -2
- package/dist/types/utils/stream-repetition-guard.d.ts +107 -0
- package/dist/types/utils/tool-call-healing.d.ts +4 -0
- package/dist/types/utils/tool-fence-strip.d.ts +27 -0
- package/package.json +3 -3
- package/src/auth-gateway/server.ts +48 -9
- package/src/auth-storage.ts +185 -48
- package/src/model-manager.ts +11 -8
- package/src/model-pricing.ts +22 -0
- package/src/model-thinking.d.ts +10 -6
- package/src/model-thinking.ts +93 -11
- package/src/models.json +241 -15
- package/src/provider-models/openai-compat.ts +27 -19
- package/src/providers/anthropic.d.ts +1 -1
- package/src/providers/anthropic.ts +10 -2
- package/src/providers/cursor.d.ts +10 -0
- package/src/providers/cursor.ts +176 -31
- package/src/providers/openai-completions.d.ts +9 -1
- package/src/providers/openai-completions.ts +379 -128
- package/src/providers/openai-opencodex-responses.ts +15 -5
- package/src/stream.ts +24 -1
- package/src/types.d.ts +16 -0
- package/src/types.ts +17 -0
- package/src/utils/discovery/openai-compatible.ts +16 -2
- package/src/utils/fallback-transport.d.ts +4 -0
- package/src/utils/fallback-transport.ts +11 -0
- package/src/utils/h2-fetch.ts +70 -7
- package/src/utils/http-inspector.ts +4 -2
- package/src/utils/idle-iterator.ts +109 -96
- package/src/utils/json-parse.ts +12 -4
- package/src/utils/stream-repetition-guard.d.ts +107 -0
- package/src/utils/stream-repetition-guard.ts +290 -0
- package/src/utils/tool-call-healing.d.ts +4 -0
- package/src/utils/tool-call-healing.ts +4 -0
- package/src/utils/tool-fence-strip.d.ts +27 -0
- package/src/utils/tool-fence-strip.ts +64 -0
|
@@ -18,6 +18,7 @@ import {
|
|
|
18
18
|
} from "../adapter-internals/provider-safety-stop";
|
|
19
19
|
import {
|
|
20
20
|
type Effort,
|
|
21
|
+
getMiniMaxThinkingMode,
|
|
21
22
|
getSupportedEfforts,
|
|
22
23
|
isGroqCompoundReasoningUnsupported,
|
|
23
24
|
modelSupportsReasoningControl,
|
|
@@ -33,6 +34,7 @@ import {
|
|
|
33
34
|
type Model,
|
|
34
35
|
type OpenAICompat,
|
|
35
36
|
type ProviderSessionState,
|
|
37
|
+
type RepetitionGuardOptions,
|
|
36
38
|
resolveServiceTier,
|
|
37
39
|
type ServiceTier,
|
|
38
40
|
type StopReason,
|
|
@@ -78,6 +80,13 @@ import { callWithCopilotModelRetry } from "../utils/retry";
|
|
|
78
80
|
import { resolveRetryBudget } from "../utils/retry-budget";
|
|
79
81
|
import { adaptSchemaForStrict, flattenToolRootCombinators, NO_STRICT, toolWireSchema } from "../utils/schema";
|
|
80
82
|
import { wrapFetchForSseDebug } from "../utils/sse-debug";
|
|
83
|
+
import {
|
|
84
|
+
DEFAULT_REPETITION_THRESHOLD,
|
|
85
|
+
REPETITION_GUARD_ERROR_CODE,
|
|
86
|
+
REPETITION_GUARD_STOP_MESSAGE,
|
|
87
|
+
StreamRepetitionGuard,
|
|
88
|
+
type StreamRepetitionTrip,
|
|
89
|
+
} from "../utils/stream-repetition-guard";
|
|
81
90
|
import { type HealedToolCall, modelMayLeakKimiToolCalls, ToolCallHealer } from "../utils/tool-call-healing";
|
|
82
91
|
import { isForcedToolChoice, mapToOpenAICompletionsToolChoice } from "../utils/tool-choice";
|
|
83
92
|
import {
|
|
@@ -85,6 +94,7 @@ import {
|
|
|
85
94
|
markToolChoiceIncapability,
|
|
86
95
|
resolveToolChoice,
|
|
87
96
|
} from "../utils/tool-choice-capability";
|
|
97
|
+
import { ToolFenceStripper } from "../utils/tool-fence-strip";
|
|
88
98
|
import { COMPOSER_EDIT_DISCIPLINE_PROMPT, isComposerHarnessModel } from "./composer-discipline";
|
|
89
99
|
import { mergeDashScopeTokenPlanHeaders } from "./dashscope-token-plan-headers";
|
|
90
100
|
import {
|
|
@@ -358,13 +368,21 @@ export interface OpenAICompletionsOptions extends StreamOptions {
|
|
|
358
368
|
/** Force-disable reasoning where supported, or request the lowest effort on generic effort endpoints. */
|
|
359
369
|
disableReasoning?: boolean;
|
|
360
370
|
serviceTier?: ServiceTier;
|
|
371
|
+
/**
|
|
372
|
+
* Runaway-repetition guard thresholds, per stream channel. A number sets the
|
|
373
|
+
* consecutive-repeat threshold; `false` disables the channel's guard.
|
|
374
|
+
* Defaults: thinking = DEFAULT_REPETITION_THRESHOLD, text = false — visible
|
|
375
|
+
* output is a deliverable and intentional repetition there (logs, fixtures,
|
|
376
|
+
* tables, generated code) must survive byte for byte (#5627).
|
|
377
|
+
*/
|
|
378
|
+
repetitionGuard?: RepetitionGuardOptions;
|
|
361
379
|
}
|
|
362
380
|
|
|
363
381
|
type OpenAICompletionsParams = Omit<OpenAI.Chat.Completions.ChatCompletionCreateParamsStreaming, "reasoning_effort"> & {
|
|
364
382
|
top_k?: number;
|
|
365
383
|
min_p?: number;
|
|
366
384
|
repetition_penalty?: number;
|
|
367
|
-
thinking?: { type: "enabled" | "disabled" };
|
|
385
|
+
thinking?: { type: "enabled" | "disabled" | "adaptive" };
|
|
368
386
|
enable_thinking?: boolean;
|
|
369
387
|
chat_template_kwargs?: { enable_thinking: boolean };
|
|
370
388
|
reasoning?: { effort?: string } | { enabled: false };
|
|
@@ -529,6 +547,20 @@ const OPENAI_COMPLETIONS_EMPTY_RESPONSE_MESSAGE = "Provider returned an empty re
|
|
|
529
547
|
const OPENAI_COMPLETIONS_NETWORK_ERROR_RETRY_MAX_RETRIES = 3;
|
|
530
548
|
const OPENAI_COMPLETIONS_NETWORK_ERROR_RETRY_BASE_DELAY_MS = 2000;
|
|
531
549
|
|
|
550
|
+
// A tripped repetition guard stops *emitting* immediately, so the user-visible
|
|
551
|
+
// symptom is already fixed at the trip. Aborting the stream right then would
|
|
552
|
+
// also drop `tool_calls` frames a provider emits *after* the repeats, losing a
|
|
553
|
+
// valid invocation (#5627). The stream is drained for a bounded window instead;
|
|
554
|
+
// the only thing the abort still buys is not burning provider budget, and that
|
|
555
|
+
// can wait this long.
|
|
556
|
+
const REPETITION_DRAIN_MAX_CHUNKS = 64;
|
|
557
|
+
const REPETITION_DRAIN_MAX_MS = 2_000;
|
|
558
|
+
// A tool call whose accumulated arguments are not yet complete JSON is worth
|
|
559
|
+
// waiting longer for — but not forever, or a call whose arguments never
|
|
560
|
+
// complete would hold the stream open for the rest of the turn's budget.
|
|
561
|
+
const REPETITION_DRAIN_PENDING_TOOL_MAX_CHUNKS = 256;
|
|
562
|
+
const REPETITION_DRAIN_PENDING_TOOL_MAX_MS = 8_000;
|
|
563
|
+
|
|
532
564
|
function hasReplayUnsafeOpenAICompletionsDelta(chunk: ChatCompletionChunk): boolean {
|
|
533
565
|
const choice = Array.isArray(chunk.choices) ? chunk.choices[0] : undefined;
|
|
534
566
|
const delta = choice?.delta;
|
|
@@ -583,6 +615,48 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
|
|
|
583
615
|
const abortTracker = createAbortSourceTracker(options?.signal);
|
|
584
616
|
const { requestAbortController, requestSignal } = abortTracker;
|
|
585
617
|
|
|
618
|
+
// Declared outside the try so the catch block — which is where the abort
|
|
619
|
+
// below lands — can tell a runaway-repetition stop from a transport error.
|
|
620
|
+
let repetitionTrip: (StreamRepetitionTrip & { channel: "text" | "thinking" }) | undefined;
|
|
621
|
+
// Drain-window bookkeeping: when the trip happened, and how many chunks have
|
|
622
|
+
// been consumed since. Both are only meaningful once `repetitionTrip` is set.
|
|
623
|
+
let repetitionTrippedAt: number | undefined;
|
|
624
|
+
let repetitionDrainedChunks = 0;
|
|
625
|
+
// Set immediately before the guard's own abort and nowhere else. The catch
|
|
626
|
+
// block must be able to tell OUR abort from a provider stall or a transport
|
|
627
|
+
// failure that merely happened to land inside the drain window: `repetitionTrip`
|
|
628
|
+
// alone is true for all three, and using it there discarded the real
|
|
629
|
+
// timeout/transport facts and flipped the retry classification (#5627 review r4).
|
|
630
|
+
let repetitionSelfAbort = false;
|
|
631
|
+
const finalizeRepetitionGuardStop = (): void => {
|
|
632
|
+
if (!repetitionTrip) return;
|
|
633
|
+
// `error`, not `aborted`: this is a provider-side failure we detected
|
|
634
|
+
// locally, and `aborted` is the wire for *client cancellation* — the
|
|
635
|
+
// auth gateway maps it to 499/`request_aborted` and telemetry counts it
|
|
636
|
+
// as a user cancel, so borrowing it misreports the turn (#5627). No new
|
|
637
|
+
// StopReason variant: the union is switched on exhaustively everywhere.
|
|
638
|
+
// `errorCode` stays the bounded classifier for *why* (#5624).
|
|
639
|
+
output.stopReason = "error";
|
|
640
|
+
output.errorCode = REPETITION_GUARD_ERROR_CODE;
|
|
641
|
+
// A fixed literal, never the observed sample/channel/count: the auth
|
|
642
|
+
// gateway forwards `errorMessage` to API clients on the streaming path,
|
|
643
|
+
// so interpolating here publishes raw model output and feeds it to a
|
|
644
|
+
// keyword classifier that picks HTTP status from message text. The
|
|
645
|
+
// diagnostic detail lives in the `logger.debug` at the trip site
|
|
646
|
+
// instead (#5627 review r5).
|
|
647
|
+
output.errorMessage = REPETITION_GUARD_STOP_MESSAGE;
|
|
648
|
+
output.duration = Date.now() - startTime;
|
|
649
|
+
if (firstTokenTime) output.ttft = firstTokenTime - startTime;
|
|
650
|
+
// No `transportFailure`: this is a local decision, not a retryable
|
|
651
|
+
// transport fault, and the agent loop's retry admission keys on that
|
|
652
|
+
// field. The session-layer classifier does not — absent transport facts
|
|
653
|
+
// it defaults to a bounded retry — so it branches on this `errorCode`
|
|
654
|
+
// and treats the trip as terminal instead (#5627). Retrying is pointless
|
|
655
|
+
// anyway: a decode loop is deterministic for the submitted context.
|
|
656
|
+
stream.push({ type: "error", reason: "error", error: output });
|
|
657
|
+
stream.end();
|
|
658
|
+
};
|
|
659
|
+
|
|
586
660
|
try {
|
|
587
661
|
const apiKey = options?.apiKey || getEnvApiKey(model.provider) || "";
|
|
588
662
|
const idleTimeoutMs = options?.streamIdleTimeoutMs ?? getOpenAIStreamIdleTimeoutMs(model.provider, model.id);
|
|
@@ -869,15 +943,117 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
|
|
|
869
943
|
|
|
870
944
|
let taggedTextBuffer = "";
|
|
871
945
|
let insideTaggedThinking = false;
|
|
946
|
+
// One guard per channel: interleaving visible text and reasoning through
|
|
947
|
+
// a single instance would splice unrelated tokens into the same window.
|
|
948
|
+
// Tool-call frames are never fed through either guard.
|
|
949
|
+
//
|
|
950
|
+
// A disabled channel gets no guard at all rather than a lenient one, so
|
|
951
|
+
// it is structurally impossible for it to set `repetitionTrip`. Visible
|
|
952
|
+
// text is disabled by default: a decode loop there is not the reported
|
|
953
|
+
// failure (#5624 was reasoning-channel), and truncating deliverable
|
|
954
|
+
// output — a log dump, a fixture, a table — corrupts the answer (#5627).
|
|
955
|
+
const createRepetitionGuard = (
|
|
956
|
+
setting: number | false | undefined,
|
|
957
|
+
fallback: number | false,
|
|
958
|
+
): StreamRepetitionGuard | undefined => {
|
|
959
|
+
const threshold = setting ?? fallback;
|
|
960
|
+
return threshold === false ? undefined : new StreamRepetitionGuard({ threshold });
|
|
961
|
+
};
|
|
962
|
+
const textRepetitionGuard = createRepetitionGuard(options?.repetitionGuard?.text, false);
|
|
963
|
+
const thinkingRepetitionGuard = createRepetitionGuard(
|
|
964
|
+
options?.repetitionGuard?.thinking,
|
|
965
|
+
DEFAULT_REPETITION_THRESHOLD,
|
|
966
|
+
);
|
|
967
|
+
const noteRepetitionTrip = (guard: StreamRepetitionGuard, channel: "text" | "thinking") => {
|
|
968
|
+
// `takeTrip()` latches once per guard; this latches once per request,
|
|
969
|
+
// so the drain window below opens exactly once no matter which
|
|
970
|
+
// channel loops.
|
|
971
|
+
if (repetitionTrip) return;
|
|
972
|
+
const trip = guard.takeTrip();
|
|
973
|
+
if (!trip) return;
|
|
974
|
+
repetitionTrip = { ...trip, channel };
|
|
975
|
+
// The repeated unit is never logged. `logger`'s default transport is a
|
|
976
|
+
// rotating file under `~/.gjc/logs` and `makeLogFormat` JSON-stringifies
|
|
977
|
+
// every metadata key verbatim — no redaction — so a sample would persist
|
|
978
|
+
// raw model output to disk and carry it into log rotation, support
|
|
979
|
+
// bundles and backups. If the loop swallowed a secret or a private
|
|
980
|
+
// fragment of the prompt, that is where it would land (#5627 review r6).
|
|
981
|
+
//
|
|
982
|
+
// Only bounded metadata the model cannot control the *content* of goes
|
|
983
|
+
// out: an id, two enums and two counts. `sampleLength` is deliberately a
|
|
984
|
+
// number, not a hash — a hash of a short secret is a probe oracle and
|
|
985
|
+
// buys nothing for debugging a decode loop. `trip.sample` itself stays on
|
|
986
|
+
// the in-memory trip for callers; this is only the logging contract.
|
|
987
|
+
//
|
|
988
|
+
// Separately, `errorMessage` stays a fixed literal because the gateway
|
|
989
|
+
// forwards it to API clients (#5627 review r5). Both hold at once.
|
|
990
|
+
logger.debug("openai-completions: repetition guard tripped", {
|
|
991
|
+
model: model.id,
|
|
992
|
+
channel,
|
|
993
|
+
kind: trip.kind,
|
|
994
|
+
repeats: trip.repeats,
|
|
995
|
+
sampleLength: trip.sample.length,
|
|
996
|
+
});
|
|
997
|
+
// Deliberately no abort here — see the REPETITION_DRAIN_* constants.
|
|
998
|
+
// The main loop closes the window once late tool-call frames have had
|
|
999
|
+
// their chance to land.
|
|
1000
|
+
repetitionTrippedAt = Date.now();
|
|
1001
|
+
repetitionDrainedChunks = 0;
|
|
1002
|
+
};
|
|
1003
|
+
/**
|
|
1004
|
+
* Closes the post-trip drain window. Called once per consumed chunk after
|
|
1005
|
+
* that chunk is fully processed, so the frames it carried are finalized
|
|
1006
|
+
* before the stream is cut.
|
|
1007
|
+
*/
|
|
1008
|
+
const maybeAbortAfterRepetitionDrain = (): void => {
|
|
1009
|
+
if (repetitionTrippedAt === undefined || requestSignal.aborted) return;
|
|
1010
|
+
// Reuses the `stopReason === "length"` truncation check below: an open
|
|
1011
|
+
// tool call whose `partialArgs` will not parse is still mid-flight.
|
|
1012
|
+
const toolCallPending =
|
|
1013
|
+
currentBlock?.type === "toolCall" &&
|
|
1014
|
+
!isCompleteJson((currentBlock as { partialArgs?: string }).partialArgs);
|
|
1015
|
+
const maxChunks = toolCallPending ? REPETITION_DRAIN_PENDING_TOOL_MAX_CHUNKS : REPETITION_DRAIN_MAX_CHUNKS;
|
|
1016
|
+
const maxMs = toolCallPending ? REPETITION_DRAIN_PENDING_TOOL_MAX_MS : REPETITION_DRAIN_MAX_MS;
|
|
1017
|
+
if (repetitionDrainedChunks >= maxChunks || Date.now() - repetitionTrippedAt >= maxMs) {
|
|
1018
|
+
repetitionSelfAbort = true;
|
|
1019
|
+
requestAbortController.abort();
|
|
1020
|
+
}
|
|
1021
|
+
};
|
|
1022
|
+
|
|
1023
|
+
// Reasoning-channel only — a fence token in visible prose must survive
|
|
1024
|
+
// as text (CHANGELOG.md:1094), and the Kimi healer must not see this
|
|
1025
|
+
// channel at all or its holdback buffer corrupts.
|
|
1026
|
+
const thinkingFenceStripper = new ToolFenceStripper();
|
|
1027
|
+
let lastThinkingSignature: string | undefined;
|
|
1028
|
+
|
|
1029
|
+
/** Returns the portion safe to emit — the whole chunk when the channel is unguarded. */
|
|
1030
|
+
const feedRepetitionGuard = (
|
|
1031
|
+
guard: StreamRepetitionGuard | undefined,
|
|
1032
|
+
text: string,
|
|
1033
|
+
channel: "text" | "thinking",
|
|
1034
|
+
): string => {
|
|
1035
|
+
if (!guard) return text;
|
|
1036
|
+
const emit = guard.feed(text);
|
|
1037
|
+
noteRepetitionTrip(guard, channel);
|
|
1038
|
+
return emit;
|
|
1039
|
+
};
|
|
1040
|
+
|
|
872
1041
|
const appendTextDelta = (text: string) => {
|
|
873
1042
|
if (!text) return;
|
|
874
1043
|
if (!firstTokenTime) firstTokenTime = Date.now();
|
|
875
|
-
|
|
1044
|
+
const emit = feedRepetitionGuard(textRepetitionGuard, text, "text");
|
|
1045
|
+
if (emit) appendText(output, stream, emit);
|
|
1046
|
+
};
|
|
1047
|
+
const emitThinkingText = (thinking: string, signature?: string) => {
|
|
1048
|
+
if (!thinking) return;
|
|
1049
|
+
const emit = feedRepetitionGuard(thinkingRepetitionGuard, thinking, "thinking");
|
|
1050
|
+
if (emit) appendThinking(output, stream, emit, signature);
|
|
876
1051
|
};
|
|
877
1052
|
const appendThinkingDelta = (thinking: string, signature?: string) => {
|
|
878
1053
|
if (!thinking) return;
|
|
879
1054
|
if (!firstTokenTime) firstTokenTime = Date.now();
|
|
880
|
-
|
|
1055
|
+
lastThinkingSignature = signature;
|
|
1056
|
+
emitThinkingText(thinkingFenceStripper.feed(thinking), signature);
|
|
881
1057
|
};
|
|
882
1058
|
|
|
883
1059
|
const flushTaggedTextBuffer = () => {
|
|
@@ -1000,150 +1176,180 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
|
|
|
1000
1176
|
};
|
|
1001
1177
|
|
|
1002
1178
|
for await (const chunk of iterateWithNetworkErrorRetry()) {
|
|
1003
|
-
|
|
1004
|
-
|
|
1005
|
-
|
|
1006
|
-
|
|
1007
|
-
|
|
1008
|
-
|
|
1009
|
-
|
|
1010
|
-
|
|
1011
|
-
|
|
1012
|
-
|
|
1013
|
-
|
|
1014
|
-
|
|
1015
|
-
|
|
1016
|
-
|
|
1017
|
-
|
|
1018
|
-
|
|
1019
|
-
applyUsage(choiceUsage);
|
|
1179
|
+
// Counted before the guarded blocks below so chunks carrying no
|
|
1180
|
+
// delta still spend the drain budget rather than extending it.
|
|
1181
|
+
if (repetitionTrippedAt !== undefined) repetitionDrainedChunks += 1;
|
|
1182
|
+
|
|
1183
|
+
// Positive-form guards instead of early `continue`s: the drain check
|
|
1184
|
+
// at the bottom of the body has to be reached on *every* path, or a
|
|
1185
|
+
// provider that keeps emitting usage-only, keepalive-shaped,
|
|
1186
|
+
// `choices`-less or malformed chunks after a trip never spends the
|
|
1187
|
+
// budget and the request hangs open (#5627 review r5).
|
|
1188
|
+
if (chunk && typeof chunk === "object") {
|
|
1189
|
+
// OpenAI documents ChatCompletionChunk.id as the unique chat completion identifier,
|
|
1190
|
+
// and each chunk in a streamed completion carries the same id.
|
|
1191
|
+
output.responseId ||= chunk.id;
|
|
1192
|
+
|
|
1193
|
+
if (chunk.usage) {
|
|
1194
|
+
applyUsage(chunk.usage);
|
|
1020
1195
|
}
|
|
1021
|
-
}
|
|
1022
1196
|
|
|
1023
|
-
|
|
1024
|
-
|
|
1025
|
-
|
|
1026
|
-
|
|
1027
|
-
|
|
1028
|
-
|
|
1029
|
-
|
|
1030
|
-
output.errorMessage = finishReasonResult.errorMessage;
|
|
1197
|
+
const choice = Array.isArray(chunk.choices) ? chunk.choices[0] : undefined;
|
|
1198
|
+
if (choice) {
|
|
1199
|
+
if (!chunk.usage) {
|
|
1200
|
+
const choiceUsage = getChoiceUsage(choice);
|
|
1201
|
+
if (choiceUsage) {
|
|
1202
|
+
applyUsage(choiceUsage);
|
|
1203
|
+
}
|
|
1031
1204
|
}
|
|
1032
|
-
}
|
|
1033
|
-
}
|
|
1034
1205
|
|
|
1035
|
-
|
|
1036
|
-
|
|
1037
|
-
|
|
1038
|
-
|
|
1039
|
-
|
|
1040
|
-
|
|
1041
|
-
|
|
1042
|
-
|
|
1043
|
-
|
|
1044
|
-
if (!firstTokenTime) firstTokenTime = Date.now();
|
|
1045
|
-
if (parseMiniMaxThinkTags) {
|
|
1046
|
-
taggedTextBuffer += normalizedDeltaText;
|
|
1047
|
-
flushTaggedTextBuffer();
|
|
1048
|
-
} else if (stripDeepseekChatTemplateTokens) {
|
|
1049
|
-
deepseekStripBuffer += normalizedDeltaText;
|
|
1050
|
-
flushDeepseekStripBuffer(false);
|
|
1051
|
-
} else if (kimiHealer) {
|
|
1052
|
-
const hasStructuredToolCalls =
|
|
1053
|
-
Array.isArray(choice.delta.tool_calls) && choice.delta.tool_calls.length > 0;
|
|
1054
|
-
if (hasStructuredToolCalls) {
|
|
1055
|
-
// Same chunk leaks markers AND carries structured tool_calls.
|
|
1056
|
-
// Strip the marker text from visible output, but drop any
|
|
1057
|
-
// synthesized calls so the structured payload stays the
|
|
1058
|
-
// single source of truth (avoids double-dispatch).
|
|
1059
|
-
const clean = kimiHealer.consumeWithoutCalls(normalizedDeltaText);
|
|
1060
|
-
if (clean.length > 0) appendTextDelta(clean);
|
|
1061
|
-
} else {
|
|
1062
|
-
const clean = kimiHealer.feed(normalizedDeltaText);
|
|
1063
|
-
if (clean.length > 0) appendTextDelta(clean);
|
|
1064
|
-
flushHealedToolCalls();
|
|
1206
|
+
if (choice.finish_reason) {
|
|
1207
|
+
const finishReasonResult = mapStopReason(choice.finish_reason);
|
|
1208
|
+
if (choice.finish_reason === "content_filter") {
|
|
1209
|
+
markProviderSafetyStop(finishReasonResult.errorMessage);
|
|
1210
|
+
} else if (!providerSafetyStop) {
|
|
1211
|
+
output.stopReason = finishReasonResult.stopReason;
|
|
1212
|
+
if (finishReasonResult.errorMessage) {
|
|
1213
|
+
output.errorMessage = finishReasonResult.errorMessage;
|
|
1214
|
+
}
|
|
1065
1215
|
}
|
|
1066
|
-
} else {
|
|
1067
|
-
appendTextDelta(normalizedDeltaText);
|
|
1068
1216
|
}
|
|
1069
|
-
}
|
|
1070
1217
|
|
|
1071
|
-
|
|
1072
|
-
|
|
1073
|
-
|
|
1074
|
-
|
|
1075
|
-
|
|
1076
|
-
|
|
1077
|
-
|
|
1078
|
-
|
|
1079
|
-
(
|
|
1080
|
-
|
|
1081
|
-
|
|
1082
|
-
|
|
1083
|
-
|
|
1084
|
-
|
|
1085
|
-
|
|
1218
|
+
if (choice.delta) {
|
|
1219
|
+
if (typeof choice.delta.refusal === "string" && choice.delta.refusal.length > 0) {
|
|
1220
|
+
appendTextDelta(choice.delta.refusal);
|
|
1221
|
+
if (!providerSafetyStop) {
|
|
1222
|
+
markProviderSafetyStop("Provider returned a safety refusal");
|
|
1223
|
+
}
|
|
1224
|
+
}
|
|
1225
|
+
const normalizedDeltaText = normalizeStreamingContentText(choice.delta.content);
|
|
1226
|
+
if (normalizedDeltaText.length > 0) {
|
|
1227
|
+
if (!firstTokenTime) firstTokenTime = Date.now();
|
|
1228
|
+
if (parseMiniMaxThinkTags) {
|
|
1229
|
+
taggedTextBuffer += normalizedDeltaText;
|
|
1230
|
+
flushTaggedTextBuffer();
|
|
1231
|
+
} else if (stripDeepseekChatTemplateTokens) {
|
|
1232
|
+
deepseekStripBuffer += normalizedDeltaText;
|
|
1233
|
+
flushDeepseekStripBuffer(false);
|
|
1234
|
+
} else if (kimiHealer) {
|
|
1235
|
+
const hasStructuredToolCalls =
|
|
1236
|
+
Array.isArray(choice.delta.tool_calls) && choice.delta.tool_calls.length > 0;
|
|
1237
|
+
if (hasStructuredToolCalls) {
|
|
1238
|
+
// Same chunk leaks markers AND carries structured tool_calls.
|
|
1239
|
+
// Strip the marker text from visible output, but drop any
|
|
1240
|
+
// synthesized calls so the structured payload stays the
|
|
1241
|
+
// single source of truth (avoids double-dispatch).
|
|
1242
|
+
const clean = kimiHealer.consumeWithoutCalls(normalizedDeltaText);
|
|
1243
|
+
if (clean.length > 0) appendTextDelta(clean);
|
|
1244
|
+
} else {
|
|
1245
|
+
const clean = kimiHealer.feed(normalizedDeltaText);
|
|
1246
|
+
if (clean.length > 0) appendTextDelta(clean);
|
|
1247
|
+
flushHealedToolCalls();
|
|
1248
|
+
}
|
|
1249
|
+
} else {
|
|
1250
|
+
appendTextDelta(normalizedDeltaText);
|
|
1251
|
+
}
|
|
1086
1252
|
}
|
|
1087
|
-
}
|
|
1088
|
-
}
|
|
1089
1253
|
|
|
1090
|
-
|
|
1091
|
-
|
|
1092
|
-
|
|
1093
|
-
|
|
1254
|
+
// Some endpoints return reasoning in reasoning_content (llama.cpp),
|
|
1255
|
+
// or reasoning (other openai compatible endpoints)
|
|
1256
|
+
// Use the first non-empty reasoning field to avoid duplication
|
|
1257
|
+
// (e.g., chutes.ai returns both reasoning_content and reasoning with same content)
|
|
1258
|
+
const reasoningFields = ["reasoning_content", "reasoning", "reasoning_text"];
|
|
1259
|
+
let foundReasoningField: string | null = null;
|
|
1260
|
+
for (const field of reasoningFields) {
|
|
1261
|
+
if (
|
|
1262
|
+
(choice.delta as any)[field] !== null &&
|
|
1263
|
+
(choice.delta as any)[field] !== undefined &&
|
|
1264
|
+
(choice.delta as any)[field].length > 0
|
|
1265
|
+
) {
|
|
1266
|
+
if (!foundReasoningField) {
|
|
1267
|
+
foundReasoningField = field;
|
|
1268
|
+
break;
|
|
1269
|
+
}
|
|
1270
|
+
}
|
|
1271
|
+
}
|
|
1094
1272
|
|
|
1095
|
-
|
|
1096
|
-
|
|
1097
|
-
|
|
1098
|
-
finishCurrentBlock(currentBlock);
|
|
1099
|
-
currentBlock = {
|
|
1100
|
-
type: "toolCall",
|
|
1101
|
-
id: toolCall.id || "",
|
|
1102
|
-
name: toolCall.function?.name || "",
|
|
1103
|
-
arguments: {},
|
|
1104
|
-
partialArgs: "",
|
|
1105
|
-
};
|
|
1106
|
-
output.content.push(currentBlock);
|
|
1107
|
-
stream.push({
|
|
1108
|
-
type: "toolcall_start",
|
|
1109
|
-
contentIndex: blockIndex(currentBlock),
|
|
1110
|
-
partial: output,
|
|
1111
|
-
});
|
|
1273
|
+
if (foundReasoningField) {
|
|
1274
|
+
const delta = (choice.delta as any)[foundReasoningField];
|
|
1275
|
+
appendThinkingDelta(delta, foundReasoningField);
|
|
1112
1276
|
}
|
|
1113
1277
|
|
|
1114
|
-
if (
|
|
1115
|
-
|
|
1116
|
-
|
|
1117
|
-
|
|
1118
|
-
|
|
1119
|
-
|
|
1120
|
-
|
|
1121
|
-
|
|
1278
|
+
if (choice?.delta?.tool_calls && choice.delta.tool_calls.length > 0) {
|
|
1279
|
+
for (const toolCall of choice.delta.tool_calls) {
|
|
1280
|
+
if (currentBlock?.type !== "toolCall" || (toolCall.id && currentBlock.id !== toolCall.id)) {
|
|
1281
|
+
finishCurrentBlock(currentBlock);
|
|
1282
|
+
currentBlock = {
|
|
1283
|
+
type: "toolCall",
|
|
1284
|
+
id: toolCall.id || "",
|
|
1285
|
+
name: toolCall.function?.name || "",
|
|
1286
|
+
arguments: {},
|
|
1287
|
+
partialArgs: "",
|
|
1288
|
+
};
|
|
1289
|
+
output.content.push(currentBlock);
|
|
1290
|
+
stream.push({
|
|
1291
|
+
type: "toolcall_start",
|
|
1292
|
+
contentIndex: blockIndex(currentBlock),
|
|
1293
|
+
partial: output,
|
|
1294
|
+
});
|
|
1295
|
+
}
|
|
1296
|
+
|
|
1297
|
+
if (currentBlock.type === "toolCall") {
|
|
1298
|
+
if (toolCall.id) currentBlock.id = toolCall.id;
|
|
1299
|
+
if (toolCall.function?.name) currentBlock.name = toolCall.function.name;
|
|
1300
|
+
let delta = "";
|
|
1301
|
+
if (toolCall.function?.arguments) {
|
|
1302
|
+
delta = toolCall.function.arguments;
|
|
1303
|
+
currentBlock.partialArgs += toolCall.function.arguments;
|
|
1304
|
+
currentBlock.arguments = parseStreamingJson(currentBlock.partialArgs);
|
|
1305
|
+
}
|
|
1306
|
+
stream.push({
|
|
1307
|
+
type: "toolcall_delta",
|
|
1308
|
+
contentIndex: blockIndex(currentBlock),
|
|
1309
|
+
delta,
|
|
1310
|
+
partial: output,
|
|
1311
|
+
});
|
|
1312
|
+
}
|
|
1122
1313
|
}
|
|
1123
|
-
stream.push({
|
|
1124
|
-
type: "toolcall_delta",
|
|
1125
|
-
contentIndex: blockIndex(currentBlock),
|
|
1126
|
-
delta,
|
|
1127
|
-
partial: output,
|
|
1128
|
-
});
|
|
1129
1314
|
}
|
|
1130
|
-
}
|
|
1131
|
-
}
|
|
1132
1315
|
|
|
1133
|
-
|
|
1134
|
-
|
|
1135
|
-
|
|
1136
|
-
|
|
1137
|
-
|
|
1138
|
-
|
|
1139
|
-
|
|
1140
|
-
|
|
1141
|
-
|
|
1316
|
+
const reasoningDetails = (choice.delta as any).reasoning_details;
|
|
1317
|
+
if (reasoningDetails && Array.isArray(reasoningDetails)) {
|
|
1318
|
+
for (const detail of reasoningDetails) {
|
|
1319
|
+
if (detail.type === "reasoning.encrypted" && detail.id && detail.data) {
|
|
1320
|
+
const matchingToolCall = output.content.find(
|
|
1321
|
+
b => b.type === "toolCall" && b.id === detail.id,
|
|
1322
|
+
) as ToolCall | undefined;
|
|
1323
|
+
if (matchingToolCall) {
|
|
1324
|
+
matchingToolCall.thoughtSignature = JSON.stringify(detail);
|
|
1325
|
+
}
|
|
1326
|
+
}
|
|
1142
1327
|
}
|
|
1143
1328
|
}
|
|
1144
1329
|
}
|
|
1145
1330
|
}
|
|
1146
1331
|
}
|
|
1332
|
+
|
|
1333
|
+
// Single invariant: exactly one evaluation per consumed chunk, on
|
|
1334
|
+
// every path. Two properties ride on this being the last *statement*
|
|
1335
|
+
// of the body rather than a `finally`:
|
|
1336
|
+
//
|
|
1337
|
+
// (a) This chunk is fully processed — anything it carried (including
|
|
1338
|
+
// `tool_calls` frames) has landed. Only now may the drain window
|
|
1339
|
+
// close and cut the stream. The check must never run *before*
|
|
1340
|
+
// the chunk's processing.
|
|
1341
|
+
// (b) A throwing chunk keeps its own transport facts. A `finally`
|
|
1342
|
+
// would also run when the body throws, and this check sets
|
|
1343
|
+
// `repetitionSelfAbort` — which the catch below branches on to
|
|
1344
|
+
// call `finalizeRepetitionGuardStop()`. A malformed payload
|
|
1345
|
+
// throwing inside the drain window would then be re-labelled as
|
|
1346
|
+
// the guard's own abort, discarding the real
|
|
1347
|
+
// errorMessage/errorStatus/transportFailure and flipping retry
|
|
1348
|
+
// admission — exactly the defect fixed in e577268e.
|
|
1349
|
+
//
|
|
1350
|
+
// Caller-abort priority is unchanged: a genuine caller abort still
|
|
1351
|
+
// wins over the guard's self-abort, and a trip is still not a cancel.
|
|
1352
|
+
maybeAbortAfterRepetitionDrain();
|
|
1147
1353
|
}
|
|
1148
1354
|
|
|
1149
1355
|
if (parseMiniMaxThinkTags && taggedTextBuffer.length > 0) {
|
|
@@ -1159,6 +1365,29 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
|
|
|
1159
1365
|
flushDeepseekStripBuffer(true);
|
|
1160
1366
|
}
|
|
1161
1367
|
|
|
1368
|
+
// A partial fence held back at the last chunk never completed, so it was
|
|
1369
|
+
// ordinary thinking text after all.
|
|
1370
|
+
emitThinkingText(thinkingFenceStripper.flush(), lastThinkingSignature);
|
|
1371
|
+
|
|
1372
|
+
// Close each guard's in-progress unit now that no more text is coming:
|
|
1373
|
+
// a final repeat with no trailing newline would otherwise go uncounted
|
|
1374
|
+
// and the runaway turn would read as a healthy completion. Must run
|
|
1375
|
+
// after the fence flush above, whose output feeds the thinking guard.
|
|
1376
|
+
//
|
|
1377
|
+
// Normal-completion path ONLY. Never finalize in the catch block: a
|
|
1378
|
+
// stream that threw mid-repeat must keep its own transport facts rather
|
|
1379
|
+
// than be reclassified as a decode loop (#5627 r4, commit c2aa25d30).
|
|
1380
|
+
// No abort either — the stream has already ended, so aborting would set
|
|
1381
|
+
// `repetitionSelfAbort` for nothing.
|
|
1382
|
+
for (const [guard, channel] of [
|
|
1383
|
+
[textRepetitionGuard, "text"],
|
|
1384
|
+
[thinkingRepetitionGuard, "thinking"],
|
|
1385
|
+
] as const) {
|
|
1386
|
+
if (!guard) continue;
|
|
1387
|
+
guard.finalize();
|
|
1388
|
+
noteRepetitionTrip(guard, channel);
|
|
1389
|
+
}
|
|
1390
|
+
|
|
1162
1391
|
if (kimiHealer) {
|
|
1163
1392
|
const trailing = kimiHealer.flushPending();
|
|
1164
1393
|
if (trailing.length > 0) appendTextDelta(trailing);
|
|
@@ -1186,6 +1415,14 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
|
|
|
1186
1415
|
|
|
1187
1416
|
finishCurrentBlock(currentBlock);
|
|
1188
1417
|
|
|
1418
|
+
// A repetition abort usually surfaces as a throw from the stream
|
|
1419
|
+
// iterator, but a host that had already buffered the rest of the
|
|
1420
|
+
// response finishes the loop normally instead. Same outcome either way.
|
|
1421
|
+
if (repetitionTrip) {
|
|
1422
|
+
finalizeRepetitionGuardStop();
|
|
1423
|
+
return;
|
|
1424
|
+
}
|
|
1425
|
+
|
|
1189
1426
|
const firstEventTimeoutError = abortTracker.getLocalAbortReason();
|
|
1190
1427
|
if (firstEventTimeoutError) {
|
|
1191
1428
|
throw firstEventTimeoutError;
|
|
@@ -1220,6 +1457,15 @@ export const streamOpenAICompletions: StreamFunction<"openai-completions"> = (
|
|
|
1220
1457
|
stream.end();
|
|
1221
1458
|
} catch (error) {
|
|
1222
1459
|
for (const block of output.content) delete (block as any).index;
|
|
1460
|
+
// Our own abort landed here. Classify it before the generic transport
|
|
1461
|
+
// path turns it into a retryable provider error. A caller abort still
|
|
1462
|
+
// wins: the user's cancel is the more meaningful intent. Keyed on the
|
|
1463
|
+
// self-abort flag, not on `repetitionTrip`: a stall or transport error
|
|
1464
|
+
// during the drain window must keep its own facts.
|
|
1465
|
+
if (repetitionSelfAbort && !abortTracker.wasCallerAbort()) {
|
|
1466
|
+
finalizeRepetitionGuardStop();
|
|
1467
|
+
return;
|
|
1468
|
+
}
|
|
1223
1469
|
const localAbortReason = abortTracker.getLocalAbortReason();
|
|
1224
1470
|
const normalizedError =
|
|
1225
1471
|
!streamConnected && model.provider === "alibaba-token-plan" && error instanceof APIConnectionTimeoutError
|
|
@@ -1576,7 +1822,12 @@ function buildParams(
|
|
|
1576
1822
|
delete params.tool_choice;
|
|
1577
1823
|
}
|
|
1578
1824
|
|
|
1579
|
-
if (supportsReasoningParams &&
|
|
1825
|
+
if (supportsReasoningParams && getMiniMaxThinkingMode(model, resolvedBaseUrl) === "toggle") {
|
|
1826
|
+
// MiniMax-M3 accepts an on/off switch, not reasoning_effort. Omitting
|
|
1827
|
+
// the switch preserves the OpenAI-compatible endpoint's default (on).
|
|
1828
|
+
if (options?.disableReasoning) params.thinking = { type: "disabled" };
|
|
1829
|
+
else if (options?.reasoning) params.thinking = { type: "adaptive" };
|
|
1830
|
+
} else if (supportsReasoningParams && compat.thinkingFormat === "zai" && model.reasoning) {
|
|
1580
1831
|
// Z.ai uses binary thinking: { type: "enabled" | "disabled" }
|
|
1581
1832
|
// Must explicitly disable since z.ai defaults to thinking enabled.
|
|
1582
1833
|
const enabled = options?.reasoning && !options?.disableReasoning;
|