@owlmeans/llm 0.1.18-rc.2 → 0.1.18-rc.21
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -2
- package/agent-meta/manifest.json +2 -2
- package/agent-meta/skills/llm/SKILL.md +239 -116
- package/agent-meta/skills/llm-prompt-caching/SKILL.md +37 -3
- package/build/execution/service.d.ts.map +1 -1
- package/build/execution/service.js +46 -11
- package/build/execution/service.js.map +1 -1
- package/build/execution/types.d.ts +72 -12
- package/build/execution/types.d.ts.map +1 -1
- package/build/execution/utils.d.ts.map +1 -1
- package/build/execution/utils.js +15 -9
- package/build/execution/utils.js.map +1 -1
- package/build/helpers/retry.d.ts +14 -0
- package/build/helpers/retry.d.ts.map +1 -1
- package/build/helpers/retry.js +14 -0
- package/build/helpers/retry.js.map +1 -1
- package/build/helpers/spectate.d.ts.map +1 -1
- package/build/helpers/spectate.js +12 -5
- package/build/helpers/spectate.js.map +1 -1
- package/build/index.d.ts +2 -2
- package/build/index.d.ts.map +1 -1
- package/build/index.js +2 -2
- package/build/index.js.map +1 -1
- package/build/model.d.ts +1 -1
- package/build/model.d.ts.map +1 -1
- package/build/model.js +60 -28
- package/build/model.js.map +1 -1
- package/build/plugins/anthropic.d.ts +40 -0
- package/build/plugins/anthropic.d.ts.map +1 -1
- package/build/plugins/anthropic.js +79 -6
- package/build/plugins/anthropic.js.map +1 -1
- package/build/plugins/openai.d.ts +12 -0
- package/build/plugins/openai.d.ts.map +1 -1
- package/build/plugins/openai.js +29 -3
- package/build/plugins/openai.js.map +1 -1
- package/build/plugins/types.d.ts +7 -0
- package/build/plugins/types.d.ts.map +1 -1
- package/build/prompt/service.d.ts.map +1 -1
- package/build/prompt/service.js +11 -0
- package/build/prompt/service.js.map +1 -1
- package/build/prompt/types.d.ts +20 -0
- package/build/prompt/types.d.ts.map +1 -1
- package/build/service.d.ts.map +1 -1
- package/build/service.js +42 -7
- package/build/service.js.map +1 -1
- package/build/types.d.ts +71 -1
- package/build/types.d.ts.map +1 -1
- package/build/utils/config.d.ts +11 -0
- package/build/utils/config.d.ts.map +1 -1
- package/build/utils/config.js +21 -1
- package/build/utils/config.js.map +1 -1
- package/build/utils/null-report.d.ts.map +1 -1
- package/build/utils/null-report.js +9 -1
- package/build/utils/null-report.js.map +1 -1
- package/package.json +13 -13
- package/src/execution/service.ts +46 -11
- package/src/execution/types.ts +77 -13
- package/src/execution/utils.ts +16 -9
- package/src/helpers/retry.ts +16 -0
- package/src/helpers/spectate.ts +14 -5
- package/src/index.ts +4 -2
- package/src/model.ts +77 -27
- package/src/plugins/anthropic.ts +89 -6
- package/src/plugins/openai.ts +33 -3
- package/src/plugins/types.ts +8 -0
- package/src/prompt/service.ts +12 -0
- package/src/prompt/types.ts +20 -0
- package/src/service.ts +53 -7
- package/src/types.ts +71 -1
- package/src/utils/config.ts +23 -1
- package/src/utils/null-report.ts +9 -1
- package/tests/context.ts +3 -0
- package/tests/execution.spec.ts +148 -13
- package/tests/helpers.spec.ts +4 -1
- package/tests/plugins.spec.ts +211 -0
- package/tests/prompt.spec.ts +43 -0
package/src/helpers/spectate.ts
CHANGED
|
@@ -5,6 +5,15 @@ import type { SpectatorEntryMessage } from '@owlmeans/llm-common'
|
|
|
5
5
|
import type { LlmSpectator, ModelInputItem } from '../types.js'
|
|
6
6
|
import { hasCacheActivity, readCacheUsage } from './cache.js'
|
|
7
7
|
|
|
8
|
+
/**
|
|
9
|
+
* The trace's `content` column is a string; LangChain content is a string-or-blocks union, and
|
|
10
|
+
* tool calls are arrays. Anything non-string must be stored as JSON — a plain string coercion
|
|
11
|
+
* downstream turns it into `[object Object]` and destroys the readable half of the trace
|
|
12
|
+
* (`raw` keeps the payload, but `content` is what trace readers query).
|
|
13
|
+
*/
|
|
14
|
+
const asText = (content: unknown): string =>
|
|
15
|
+
typeof content === 'string' ? content : JSON.stringify(content) ?? ''
|
|
16
|
+
|
|
8
17
|
/** Normalize one prompt message into the spectator's storage shape. */
|
|
9
18
|
const describeInput = (msg: ModelInputItem, callType: string): SpectatorEntryMessage => {
|
|
10
19
|
const entry: SpectatorEntryMessage = {
|
|
@@ -16,17 +25,17 @@ const describeInput = (msg: ModelInputItem, callType: string): SpectatorEntryMes
|
|
|
16
25
|
|
|
17
26
|
if (msg instanceof BaseMessage) {
|
|
18
27
|
entry.type = msg.type
|
|
19
|
-
entry.content = msg.content
|
|
28
|
+
entry.content = asText(msg.content)
|
|
20
29
|
entry.name = msg.name
|
|
21
30
|
entry.contentType = typeof msg.content === 'string' ? SpectatorContentType.Text : SpectatorContentType.Json
|
|
22
31
|
entry.usage = 'usage_metadata' in msg ? msg.usage_metadata as UsageMetadata : undefined
|
|
23
32
|
entry.raw = JSON.stringify({ message: msg, content: msg.content, meta: msg.response_metadata })
|
|
24
33
|
} else if (typeof msg === 'object' && 'content' in msg) {
|
|
25
34
|
entry.type = msg.role as string ?? 'unknown'
|
|
26
|
-
entry.content = msg.content
|
|
35
|
+
entry.content = asText(msg.content ?? msg)
|
|
27
36
|
entry.raw = JSON.stringify(msg)
|
|
28
37
|
} else {
|
|
29
|
-
entry.content = msg
|
|
38
|
+
entry.content = asText(msg)
|
|
30
39
|
entry.raw = JSON.stringify(msg)
|
|
31
40
|
}
|
|
32
41
|
|
|
@@ -58,10 +67,10 @@ export const spectate = (spectator: LlmSpectator, callType: string) =>
|
|
|
58
67
|
}
|
|
59
68
|
|
|
60
69
|
if (message.tool_calls != null && message.tool_calls.length > 0) {
|
|
61
|
-
completion.content = message.tool_calls
|
|
70
|
+
completion.content = asText(message.tool_calls)
|
|
62
71
|
completion.contentType = SpectatorContentType.ToolCall
|
|
63
72
|
} else {
|
|
64
|
-
completion.content = message.content
|
|
73
|
+
completion.content = asText(message.content)
|
|
65
74
|
}
|
|
66
75
|
|
|
67
76
|
// Silent unless the provider reported cache activity, so it costs nothing when
|
package/src/index.ts
CHANGED
|
@@ -9,6 +9,8 @@ export * from './execution/index.js'
|
|
|
9
9
|
export * from './prompt/index.js'
|
|
10
10
|
export type * from './plugins/types.js'
|
|
11
11
|
export { plugins, registerLlmPlugin, pluginOf, pluginFor, resolvePlugin } from './plugins/index.js'
|
|
12
|
-
export { anthropicPlugin, ANTHROPIC_FAMILY } from './plugins/anthropic.js'
|
|
12
|
+
export { anthropicPlugin, ANTHROPIC_FAMILY, NO_SAMPLING_PREFIXES, rejectsSampling } from './plugins/anthropic.js'
|
|
13
13
|
export { compatiblePlugin } from './plugins/compatible.js'
|
|
14
|
-
export {
|
|
14
|
+
export {
|
|
15
|
+
openAiPlugin, openAiFamily, OPENAI_FAMILY, RESPONSES_API_PREFIXES, usesResponsesApi,
|
|
16
|
+
} from './plugins/openai.js'
|
package/src/model.ts
CHANGED
|
@@ -6,7 +6,7 @@ import type { BaseChatModel } from '@langchain/core/language_models/chat_models'
|
|
|
6
6
|
import { StructuredMode } from '@owlmeans/llm-common'
|
|
7
7
|
import type { NullKind } from '@owlmeans/llm-common'
|
|
8
8
|
import {
|
|
9
|
-
|
|
9
|
+
DEFAULT_MODEL_RETRIES, FALLBACK_AFTER_ATTEMPTS, MAX_CACHE_BREAKPOINTS,
|
|
10
10
|
} from './consts.js'
|
|
11
11
|
import { LlmModelError } from './errors.js'
|
|
12
12
|
import { pluginFor, pluginOf } from './plugins/index.js'
|
|
@@ -15,7 +15,7 @@ import { coerceToSchema, parseJsonContent } from './helpers/json.js'
|
|
|
15
15
|
import { normalizeInput } from './helpers/messages.js'
|
|
16
16
|
import { withRetry } from './helpers/retry.js'
|
|
17
17
|
import { spectate } from './helpers/spectate.js'
|
|
18
|
-
import { idleTimeout, readConfig } from './utils/config.js'
|
|
18
|
+
import { idleTimeout, readConfig, resolveOutputCap } from './utils/config.js'
|
|
19
19
|
import { reportNull } from './utils/null-report.js'
|
|
20
20
|
import type { NullReportParams } from './utils/null-report.js'
|
|
21
21
|
import { applyNoThink, dropBlankContent, ensureJsonMention, stripCacheMarkers } from './utils/prompt.js'
|
|
@@ -88,6 +88,7 @@ export const makeLlmModel = ({
|
|
|
88
88
|
prompt,
|
|
89
89
|
prompts,
|
|
90
90
|
files,
|
|
91
|
+
utility,
|
|
91
92
|
}: LlmModelOptions, spectator: LlmSpectator): LlmModel => {
|
|
92
93
|
|
|
93
94
|
const ajv = new Ajv({ strict: false })
|
|
@@ -98,6 +99,17 @@ export const makeLlmModel = ({
|
|
|
98
99
|
const plugin: LlmPlugin | undefined = pluginOf(config.provider) ?? pluginFor(model)
|
|
99
100
|
const timeout = idleTimeout(config)
|
|
100
101
|
|
|
102
|
+
/**
|
|
103
|
+
* Where on the escalation ladder this call starts.
|
|
104
|
+
*
|
|
105
|
+
* The ladder has exactly `retries` rungs, so a seed past the last one buys nothing and
|
|
106
|
+
* would only inflate the attempt number handed to `refine`. Clamped, `refineModel` sees
|
|
107
|
+
* at most `2 * (retries - 1)` — the escalator's own doubling stays bounded by the
|
|
108
|
+
* output cap either way.
|
|
109
|
+
*/
|
|
110
|
+
const ladderSeed = (escalation?: number): number =>
|
|
111
|
+
Math.max(0, Math.min(Math.floor(escalation ?? 0), retries - 1))
|
|
112
|
+
|
|
101
113
|
/**
|
|
102
114
|
* Normalize, compose the system prompt, then apply every in-place prompt adaptation, in
|
|
103
115
|
* dependency order.
|
|
@@ -133,7 +145,7 @@ export const makeLlmModel = ({
|
|
|
133
145
|
callSkills: callSkills ?? prompt?.callSkills,
|
|
134
146
|
},
|
|
135
147
|
msgs,
|
|
136
|
-
{ model, provider: plugin, purpose, action, cacheMax, files },
|
|
148
|
+
{ model, provider: plugin, purpose, action, cacheMax, files, utility },
|
|
137
149
|
)
|
|
138
150
|
if (composed.system != null) {
|
|
139
151
|
msgs.unshift(composed.system)
|
|
@@ -147,7 +159,9 @@ export const makeLlmModel = ({
|
|
|
147
159
|
}
|
|
148
160
|
|
|
149
161
|
if (json) ensureJsonMention(msgs)
|
|
150
|
-
|
|
162
|
+
// The soft switch is for models with no request-level control; a plugin that sends the
|
|
163
|
+
// real parameter must not also get the directive as prompt text.
|
|
164
|
+
applyNoThink(msgs, config.disableThinking === true && plugin?.suppressesThinking?.(config) !== true)
|
|
151
165
|
// Cache markers replace string content with content blocks, so they must go last.
|
|
152
166
|
const ttl = prompt?.cacheTtl
|
|
153
167
|
const marked = plugin?.patchCache?.(msgs, {
|
|
@@ -216,9 +230,9 @@ export const makeLlmModel = ({
|
|
|
216
230
|
const basePlugin = pluginOf(baseConfig.provider) ?? pluginFor(base)
|
|
217
231
|
if (basePlugin == null) return base
|
|
218
232
|
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
233
|
+
// Read from the ACTIVE base — after the fallback swap that is the fallback's own
|
|
234
|
+
// config, so the escalator sizes the model it is actually talking to.
|
|
235
|
+
const maxOutputCap = resolveOutputCap(baseConfig)
|
|
222
236
|
const refined = basePlugin.refine({ base, attempt, temperature, maxOutputCap })
|
|
223
237
|
|
|
224
238
|
if (attempt > 0) {
|
|
@@ -297,11 +311,15 @@ export const makeLlmModel = ({
|
|
|
297
311
|
const helper: LlmModel = {
|
|
298
312
|
ask: async (
|
|
299
313
|
input,
|
|
300
|
-
{
|
|
314
|
+
{
|
|
315
|
+
ref, filter, action, useCache = false, cacheMax = MAX_CACHE_BREAKPOINTS, skills,
|
|
316
|
+
escalation, fatal,
|
|
317
|
+
}: LlmAskOptions
|
|
301
318
|
) => {
|
|
302
319
|
const msgs = await prepare(input, action, useCache, cacheMax, false, skills)
|
|
303
|
-
|
|
304
|
-
|
|
320
|
+
const seed = ladderSeed(escalation)
|
|
321
|
+
return withRetry({ retries, outputErrors, fatal }, async i => {
|
|
322
|
+
const refined = refineModel(seed + i)
|
|
305
323
|
console.log('Use model to ask: ', refined.getName(), refined.lc_kwargs.model)
|
|
306
324
|
const startedAt = Date.now()
|
|
307
325
|
let result: AIMessageChunk | null = null
|
|
@@ -315,7 +333,7 @@ export const makeLlmModel = ({
|
|
|
315
333
|
}
|
|
316
334
|
|
|
317
335
|
const message = new AIMessage(result)
|
|
318
|
-
let output: string
|
|
336
|
+
let output: string = typeof result.content === 'string'
|
|
319
337
|
? result.content
|
|
320
338
|
: Array.isArray(result.content)
|
|
321
339
|
? result.content
|
|
@@ -323,19 +341,39 @@ export const makeLlmModel = ({
|
|
|
323
341
|
typeof c === 'object' && c !== null && 'type' in c && c.type === 'text'
|
|
324
342
|
)
|
|
325
343
|
.map(c => c.text)
|
|
326
|
-
.join('')
|
|
327
|
-
:
|
|
344
|
+
.join('')
|
|
345
|
+
: ''
|
|
346
|
+
|
|
347
|
+
// A completion can carry text in blocks this strict filter does not name — the tolerant
|
|
348
|
+
// extractor reads any block with a string `text`. Only consulted once the strict pass
|
|
349
|
+
// found nothing, so the usual path keeps its exact spacing.
|
|
350
|
+
if (output.trim() === '') {
|
|
351
|
+
output = textOf(result.content)
|
|
352
|
+
}
|
|
328
353
|
|
|
329
354
|
const entry = await spectate(spectator, 'ask')(msgs, message, action, i, startedAt)
|
|
330
355
|
if (ref != null) ref.spectatorEntry = entry
|
|
331
356
|
|
|
357
|
+
// An empty completion is a NULL RESULT, and it is diagnosed here rather than blamed on
|
|
358
|
+
// the caller. Both shipped filters return null only for empty input, so letting one run
|
|
359
|
+
// first reported every empty answer as `filter-rejected` — naming the innocent party and,
|
|
360
|
+
// worse, skipping `reportNull`, whose stop reason and output-token count are the only
|
|
361
|
+
// things that say WHY nothing came back (a model that spent its whole budget thinking).
|
|
362
|
+
if (output.trim() === '') {
|
|
363
|
+
throw await nullResult('ask', {
|
|
364
|
+
action, attempt: i, startedAt, refined, msgs, raw: result, useCache,
|
|
365
|
+
})
|
|
366
|
+
}
|
|
367
|
+
|
|
332
368
|
if (filter != null) {
|
|
333
|
-
|
|
334
|
-
|
|
335
|
-
|
|
369
|
+
const produced = output
|
|
370
|
+
const filtered = await filter(produced, message)
|
|
371
|
+
if (filtered == null) {
|
|
372
|
+
// The OUTPUT, not the message envelope: `JSON.stringify(new AIMessage(...))` is 80
|
|
373
|
+
// constant characters of LangChain serialization stub and says nothing at all.
|
|
374
|
+
throw new LlmModelError(`filter-rejected:${produced.substring(0, 200)}`)
|
|
336
375
|
}
|
|
337
|
-
|
|
338
|
-
throw new LlmModelError(`empty-content:${JSON.stringify(message).substring(0, 50)}...`)
|
|
376
|
+
output = filtered
|
|
339
377
|
}
|
|
340
378
|
|
|
341
379
|
notifyRef(ref, message)
|
|
@@ -345,11 +383,15 @@ export const makeLlmModel = ({
|
|
|
345
383
|
|
|
346
384
|
talk: async (
|
|
347
385
|
input,
|
|
348
|
-
{
|
|
386
|
+
{
|
|
387
|
+
ref, filter, action, useCache = false, cacheMax = MAX_CACHE_BREAKPOINTS, skills,
|
|
388
|
+
escalation, fatal,
|
|
389
|
+
}: LlmTalkOptions
|
|
349
390
|
) => {
|
|
350
391
|
const msgs = await prepare(input, action, useCache, cacheMax, false, skills)
|
|
351
|
-
|
|
352
|
-
|
|
392
|
+
const seed = ladderSeed(escalation)
|
|
393
|
+
return withRetry({ retries, outputErrors, fatal }, async i => {
|
|
394
|
+
const refined = refineModel(seed + i)
|
|
353
395
|
console.log('Use model to talk: ', refined.getName(), refined.lc_kwargs.model)
|
|
354
396
|
const startedAt = Date.now()
|
|
355
397
|
let result: AIMessageChunk | null = null
|
|
@@ -381,14 +423,18 @@ export const makeLlmModel = ({
|
|
|
381
423
|
invoke: async <T>(
|
|
382
424
|
input: ModelInput,
|
|
383
425
|
schema: JSONSchemaType<T>,
|
|
384
|
-
{
|
|
426
|
+
{
|
|
427
|
+
temperature, ref, filter, action, useCache = false, cacheMax = MAX_CACHE_BREAKPOINTS,
|
|
428
|
+
skills, escalation, fatal,
|
|
429
|
+
}: LlmInvokeOptions<T>
|
|
385
430
|
) => {
|
|
386
431
|
const msgs = await prepare(input, action, useCache, cacheMax, true, skills)
|
|
387
432
|
const { name, innerSchema, validate } = resolveSchemaValidator<T>(ajv, schema)
|
|
388
433
|
const toolName = toToolName((innerSchema as { title?: string }).title ?? name)
|
|
389
434
|
|
|
390
|
-
|
|
391
|
-
|
|
435
|
+
const seed = ladderSeed(escalation)
|
|
436
|
+
return withRetry({ retries, outputErrors, fatal }, async i => {
|
|
437
|
+
const refined = refineModel(seed + i, temperature)
|
|
392
438
|
console.log('Use model invoke: ', refined.getName(), refined.lc_kwargs.model)
|
|
393
439
|
const startedAt = Date.now()
|
|
394
440
|
const { piece, result: collected } = await streamStructured(refined, msgs, innerSchema, toolName, action)
|
|
@@ -429,14 +475,18 @@ export const makeLlmModel = ({
|
|
|
429
475
|
request: async <T>(
|
|
430
476
|
input: ModelInput,
|
|
431
477
|
schema: JSONSchemaType<T>,
|
|
432
|
-
{
|
|
478
|
+
{
|
|
479
|
+
ref, filter, action, useCache = false, cacheMax = MAX_CACHE_BREAKPOINTS, skills,
|
|
480
|
+
escalation, fatal,
|
|
481
|
+
}: LlmRequestOptions
|
|
433
482
|
) => {
|
|
434
483
|
const msgs = await prepare(input, action, useCache, cacheMax, true, skills)
|
|
435
484
|
const { name, innerSchema, validate } = resolveSchemaValidator<T>(ajv, schema)
|
|
436
485
|
const toolName = toToolName((innerSchema as { title?: string }).title ?? name)
|
|
437
486
|
|
|
438
|
-
|
|
439
|
-
|
|
487
|
+
const seed = ladderSeed(escalation)
|
|
488
|
+
return withRetry({ retries, outputErrors, fatal }, async i => {
|
|
489
|
+
const refined = refineModel(seed + i)
|
|
440
490
|
console.log('Use model request: ', refined.getName(), refined.lc_kwargs.model)
|
|
441
491
|
const startedAt = Date.now()
|
|
442
492
|
const { piece, result: collected } = await streamStructured(refined, msgs, innerSchema, toolName, action)
|
package/src/plugins/anthropic.ts
CHANGED
|
@@ -5,13 +5,68 @@ import type { MessageContent, MessageFieldWithRole } from '@langchain/core/messa
|
|
|
5
5
|
import { ModelProvider, PromptBlock, StructuredMode } from '@owlmeans/llm-common'
|
|
6
6
|
import type { CacheTtl } from '@owlmeans/llm-common'
|
|
7
7
|
import type { LlmPlugin } from './types.js'
|
|
8
|
+
import type { ModelConfig } from '../types.js'
|
|
8
9
|
import { CHARS_PER_TOKEN, MAX_CACHE_BREAKPOINTS, MIN_CACHEABLE_TOKENS } from '../consts.js'
|
|
10
|
+
import { resolveOutputCap } from '../utils/config.js'
|
|
9
11
|
import { readConfig } from '../utils/config.js'
|
|
10
12
|
import { escalateMaxTokens, isBadRequest, makeClientOptions } from './utils.js'
|
|
11
13
|
|
|
12
14
|
/** Model-name prefix that supports prompt caching through `cache_control` markers. */
|
|
13
15
|
const CACHEABLE_PREFIX = 'claude-'
|
|
14
16
|
|
|
17
|
+
/**
|
|
18
|
+
* Model families that REJECT the sampling parameters — Claude 4.7 and later, and the whole
|
|
19
|
+
* 5 family. `temperature`, `top_p` and `top_k` were removed there, and sending any of them
|
|
20
|
+
* is a 400, not a silently ignored field. Matched with `startsWith`, so a dated snapshot
|
|
21
|
+
* (`claude-sonnet-5-20260114`) is covered by its base id.
|
|
22
|
+
*
|
|
23
|
+
* This is the Anthropic counterpart of the OpenAI plugin's `RESPONSES_API_PREFIXES`: the
|
|
24
|
+
* older models below the line (`claude-sonnet-4-6`, `claude-haiku-4-5`, and earlier) still
|
|
25
|
+
* accept sampling and still want the deterministic `temperature: 0` default.
|
|
26
|
+
*/
|
|
27
|
+
export const NO_SAMPLING_PREFIXES = [
|
|
28
|
+
'claude-fable-5',
|
|
29
|
+
'claude-mythos-5',
|
|
30
|
+
'claude-mythos-preview',
|
|
31
|
+
'claude-opus-5',
|
|
32
|
+
'claude-opus-4-8',
|
|
33
|
+
'claude-opus-4-7',
|
|
34
|
+
'claude-sonnet-5',
|
|
35
|
+
]
|
|
36
|
+
|
|
37
|
+
/** Whether this model id rejects `temperature`/`top_p`/`top_k`. */
|
|
38
|
+
export const rejectsSampling = (model: string | undefined): boolean =>
|
|
39
|
+
model != null && NO_SAMPLING_PREFIXES.some(prefix => model.startsWith(prefix))
|
|
40
|
+
|
|
41
|
+
/**
|
|
42
|
+
* Whether the request has to say, on the wire, that the model must not reason.
|
|
43
|
+
*
|
|
44
|
+
* The adaptive family reasons unless told otherwise: an absent `thinking` parameter means
|
|
45
|
+
* "adaptive", and langchain forwards the parameter only when a caller sets it — so a config
|
|
46
|
+
* that asks for no thinking is only honoured if the plugin sends `thinking: disabled` itself.
|
|
47
|
+
* Silent reasoning is what the request pays for twice: its tokens bill as output, and the
|
|
48
|
+
* summarised stream delivers them in bursts minutes apart, which an idle deadline reads as a
|
|
49
|
+
* dead connection and retries from scratch. Older models reason only when asked and get nothing.
|
|
50
|
+
*/
|
|
51
|
+
export const suppressesThinking = (config: Pick<ModelConfig, 'model' | 'disableThinking'>): boolean =>
|
|
52
|
+
config.disableThinking === true && rejectsSampling(config.model)
|
|
53
|
+
|
|
54
|
+
/**
|
|
55
|
+
* The smallest output budget an always-reasoning model is given.
|
|
56
|
+
*
|
|
57
|
+
* The same models that took the sampling knobs away also think ADAPTIVELY unless the request
|
|
58
|
+
* turns it off (`disableThinking` → `thinking: disabled`, see `suppressesThinking`), and by
|
|
59
|
+
* default that thinking is not shown — it arrives as thinking blocks with empty text. Reasoning is billed against the same `max_tokens` as the answer, so a budget
|
|
60
|
+
* sized for the answer alone can be spent entirely on thinking: the response is a well-formed
|
|
61
|
+
* completion carrying no text at all, `stop_reason: "max_tokens"`, and every retry at the same
|
|
62
|
+
* budget draws from the same distribution.
|
|
63
|
+
*
|
|
64
|
+
* The floor buys room for the reasoning AND the answer. It is a floor, not an override — a preset
|
|
65
|
+
* asking for more keeps it — and it is clamped to what the provider accepts, so it can never turn
|
|
66
|
+
* a retryable empty answer into a 400.
|
|
67
|
+
*/
|
|
68
|
+
export const ADAPTIVE_MIN_MAX_TOKENS = 32_000
|
|
69
|
+
|
|
15
70
|
export const ANTHROPIC_FAMILY = 'anthropic'
|
|
16
71
|
|
|
17
72
|
type ContentBlock = Record<string, unknown>
|
|
@@ -91,19 +146,38 @@ export const anthropicPlugin: LlmPlugin = {
|
|
|
91
146
|
*/
|
|
92
147
|
toolChoice: (toolName: string): unknown => ({ type: 'tool', name: toolName }),
|
|
93
148
|
|
|
149
|
+
suppressesThinking: config => suppressesThinking(config),
|
|
150
|
+
|
|
94
151
|
build: ({ config, secret, callbacks }) => {
|
|
95
|
-
const model = config.model ??= 'claude-haiku-4-5
|
|
152
|
+
const model = config.model ??= 'claude-haiku-4-5'
|
|
153
|
+
// Claude 4.7+ took the sampling knobs away: not "ignored", a 400. A configured
|
|
154
|
+
// `temperature` on such a model is a preset bug, and dropping it here is the only
|
|
155
|
+
// reading that keeps the call alive — there is nothing to translate it into.
|
|
156
|
+
const sampling = rejectsSampling(model)
|
|
157
|
+
? {}
|
|
158
|
+
: {
|
|
159
|
+
// Neither knob set → pin temperature to 0 for determinism.
|
|
160
|
+
...(config.temperature == null && config.topP == null ? { temperature: 0 } : {}),
|
|
161
|
+
...(config.temperature != null ? { temperature: config.temperature } : {}),
|
|
162
|
+
...(config.topP != null && config.temperature == null ? { topP: config.topP } : {}),
|
|
163
|
+
}
|
|
164
|
+
// Room for the reasoning these models always do, and for the answer after it.
|
|
165
|
+
const requested = config.maxTokens ?? 4096
|
|
166
|
+
const maxTokens = rejectsSampling(model)
|
|
167
|
+
? Math.min(Math.max(requested, ADAPTIVE_MIN_MAX_TOKENS), resolveOutputCap(config))
|
|
168
|
+
: requested
|
|
169
|
+
|
|
96
170
|
const cfg = {
|
|
97
171
|
model,
|
|
98
172
|
apiKey: secret,
|
|
99
|
-
|
|
100
|
-
...(config.temperature == null && config.topP == null ? { temperature: 0 } : {}),
|
|
101
|
-
maxTokens: config.maxTokens ?? 4096,
|
|
173
|
+
maxTokens,
|
|
102
174
|
maxRetries: 5,
|
|
103
175
|
metadata: { config },
|
|
104
176
|
callbacks,
|
|
105
|
-
...
|
|
106
|
-
...(
|
|
177
|
+
...sampling,
|
|
178
|
+
...(suppressesThinking({ model, disableThinking: config.disableThinking })
|
|
179
|
+
? { thinking: { type: 'disabled' as const } }
|
|
180
|
+
: {}),
|
|
107
181
|
...makeClientOptions({ headers: config.headers }),
|
|
108
182
|
}
|
|
109
183
|
// Anthropic rejects temperature and top_p together.
|
|
@@ -118,6 +192,15 @@ export const anthropicPlugin: LlmPlugin = {
|
|
|
118
192
|
const model = base as ChatAnthropic
|
|
119
193
|
const currentTemperature = temperature ?? model.temperature ?? 0
|
|
120
194
|
const maxTokens = escalateMaxTokens(model.maxTokens, attempt, maxOutputCap)
|
|
195
|
+
// `lc_kwargs` carries whatever `build` put there, so a no-sampling model arrives clean;
|
|
196
|
+
// what has to be suppressed is the escalator's own re-application of a temperature.
|
|
197
|
+
if (rejectsSampling(model.modelName ?? model.model)) {
|
|
198
|
+
const cfg = { ...(model.lc_kwargs as Partial<ChatAnthropic>), maxTokens }
|
|
199
|
+
delete cfg.temperature
|
|
200
|
+
delete cfg.topP
|
|
201
|
+
|
|
202
|
+
return new ChatAnthropic(cfg as Partial<ChatAnthropic>)
|
|
203
|
+
}
|
|
121
204
|
const cfg: Partial<ChatAnthropic> = {
|
|
122
205
|
...(model.lc_kwargs as Partial<ChatAnthropic>), temperature: currentTemperature, maxTokens,
|
|
123
206
|
}
|
package/src/plugins/openai.ts
CHANGED
|
@@ -5,8 +5,20 @@ import type { LlmPlugin, LlmRefineParams } from './types.js'
|
|
|
5
5
|
import type { ModelConfig } from '../types.js'
|
|
6
6
|
import { escalateMaxTokens, isBadRequest, makeConfiguration } from './utils.js'
|
|
7
7
|
|
|
8
|
-
/**
|
|
9
|
-
|
|
8
|
+
/**
|
|
9
|
+
* Model families served through OpenAI's Responses API rather than chat completions. That
|
|
10
|
+
* endpoint REJECTS `temperature`/`top_p` — a 400 naming the parameter, not a silently
|
|
11
|
+
* ignored field. Matched with `startsWith`, so a dated snapshot (`gpt-5.6-terra-2026-08`)
|
|
12
|
+
* is covered by its base id.
|
|
13
|
+
*
|
|
14
|
+
* This is the OpenAI counterpart of the anthropic plugin's `NO_SAMPLING_PREFIXES`, and it
|
|
15
|
+
* gates BOTH hooks for the same reason: see `refine`.
|
|
16
|
+
*/
|
|
17
|
+
export const RESPONSES_API_PREFIXES = ['gpt-5', 'codex-']
|
|
18
|
+
|
|
19
|
+
/** Whether this model id goes through the Responses API and therefore rejects sampling. */
|
|
20
|
+
export const usesResponsesApi = (model: string | undefined): boolean =>
|
|
21
|
+
model != null && RESPONSES_API_PREFIXES.some(prefix => model.startsWith(prefix))
|
|
10
22
|
|
|
11
23
|
export const OPENAI_FAMILY = 'openai'
|
|
12
24
|
|
|
@@ -52,6 +64,8 @@ export const openAiFamily = {
|
|
|
52
64
|
const baseKwargs = model.lc_kwargs as ConstructorParameters<typeof ChatOpenAI>[0] & {
|
|
53
65
|
modelKwargs?: { reasoning?: { max_tokens?: number } } & Record<string, unknown>
|
|
54
66
|
}
|
|
67
|
+
const responsesApi = usesResponsesApi(model.model ?? baseKwargs.model)
|
|
68
|
+
|| baseKwargs.useResponsesApi === true
|
|
55
69
|
// The dominant cause of an empty response is a reasoning model spending the whole
|
|
56
70
|
// budget on hidden thinking (finish_reason=length, empty content). The retry already
|
|
57
71
|
// raises maxTokens; ALSO shrink the absolute reasoning cap so the extra budget becomes
|
|
@@ -65,6 +79,22 @@ export const openAiFamily = {
|
|
|
65
79
|
}
|
|
66
80
|
: baseKwargs.modelKwargs
|
|
67
81
|
|
|
82
|
+
// `build` hands a Responses-API model over without sampling knobs, but EVERY call is
|
|
83
|
+
// made on the instance `refine` returns — attempt 0 included — so re-applying a
|
|
84
|
+
// temperature here puts it on the wire for every single request, not just a retry.
|
|
85
|
+
// Suppressing it in one hook and restoring it in the other ships the parameter anyway.
|
|
86
|
+
if (responsesApi) {
|
|
87
|
+
const cfg = {
|
|
88
|
+
...baseKwargs,
|
|
89
|
+
maxTokens,
|
|
90
|
+
...(modelKwargs != null ? { modelKwargs } : {}),
|
|
91
|
+
}
|
|
92
|
+
delete cfg.temperature
|
|
93
|
+
delete cfg.topP
|
|
94
|
+
|
|
95
|
+
return new ChatOpenAI(cfg)
|
|
96
|
+
}
|
|
97
|
+
|
|
68
98
|
return new ChatOpenAI({
|
|
69
99
|
...baseKwargs,
|
|
70
100
|
temperature: currentTemperature,
|
|
@@ -104,7 +134,7 @@ export const openAiPlugin: LlmPlugin = {
|
|
|
104
134
|
const modelKwargs = { prompt_cache_key: config.cacheKey ?? alias }
|
|
105
135
|
|
|
106
136
|
// The Responses API models reject `temperature`/`topP`.
|
|
107
|
-
if (
|
|
137
|
+
if (usesResponsesApi(model)) {
|
|
108
138
|
return new ChatOpenAI({
|
|
109
139
|
model,
|
|
110
140
|
apiKey: secret,
|
package/src/plugins/types.ts
CHANGED
|
@@ -93,6 +93,14 @@ export interface LlmPlugin {
|
|
|
93
93
|
*/
|
|
94
94
|
refine: (params: LlmRefineParams) => BaseChatModel
|
|
95
95
|
|
|
96
|
+
/**
|
|
97
|
+
* Does this plugin turn the model's reasoning off NATIVELY when the config asks for it
|
|
98
|
+
* (`ModelConfig.disableThinking`)? When it does, the service must not also inject the
|
|
99
|
+
* `/no_think` prompt directive — that is a soft switch for models with no request-level
|
|
100
|
+
* control, and on a provider that has one it is nothing but text in the prompt.
|
|
101
|
+
*/
|
|
102
|
+
suppressesThinking?: (config: Pick<ModelConfig, 'model' | 'disableThinking'>) => boolean
|
|
103
|
+
|
|
96
104
|
/** How this provider should be asked for schema-conforming output. */
|
|
97
105
|
structuredMode: (config: ModelConfig) => StructuredMode
|
|
98
106
|
|
package/src/prompt/service.ts
CHANGED
|
@@ -105,6 +105,10 @@ export const promptServiceApi = (
|
|
|
105
105
|
|
|
106
106
|
compose: async (input, messages, params): Promise<PromptResult> => {
|
|
107
107
|
const sections = new Map<PromptBlock, string[]>()
|
|
108
|
+
// Scoped to this composition, never to the service: a claim is about who renders a
|
|
109
|
+
// thing in ONE prompt, and carrying it across calls would silently drop the content
|
|
110
|
+
// from every later prompt that shares the service.
|
|
111
|
+
const claimed = new Set<string>()
|
|
108
112
|
const ctx: PromptContext = {
|
|
109
113
|
...params,
|
|
110
114
|
input,
|
|
@@ -122,6 +126,14 @@ export const promptServiceApi = (
|
|
|
122
126
|
}
|
|
123
127
|
},
|
|
124
128
|
resolve: aliases => self().resolve(aliases),
|
|
129
|
+
claim: key => {
|
|
130
|
+
if (claimed.has(key)) {
|
|
131
|
+
return false
|
|
132
|
+
}
|
|
133
|
+
claimed.add(key)
|
|
134
|
+
|
|
135
|
+
return true
|
|
136
|
+
},
|
|
125
137
|
}
|
|
126
138
|
|
|
127
139
|
// Two passes, not one: every static contribution must be in place before a plugin
|
package/src/prompt/types.ts
CHANGED
|
@@ -36,6 +36,16 @@ export interface PromptComposeParams {
|
|
|
36
36
|
cacheMax?: number
|
|
37
37
|
/** File access a plugin may use to resolve knowledge from disk. */
|
|
38
38
|
files?: FileProviderRef
|
|
39
|
+
/**
|
|
40
|
+
* A cheap model a plugin may spend ONE call on while composing — picking which of a
|
|
41
|
+
* hundred candidate skills a request is actually about, classifying an ask. Resolved
|
|
42
|
+
* lazily and allowed to yield `undefined`: no cheap tier is configured on most
|
|
43
|
+
* deployments, and a plugin that cannot get one must degrade rather than fail.
|
|
44
|
+
*
|
|
45
|
+
* Whatever it returns must not change what lands in a CACHED block — a model's answer
|
|
46
|
+
* is not reproducible byte-for-byte, so it belongs in `Packages` or `Context`.
|
|
47
|
+
*/
|
|
48
|
+
utility?: () => BaseChatModel | undefined
|
|
39
49
|
}
|
|
40
50
|
|
|
41
51
|
/** What a prompt plugin sees and may contribute to. */
|
|
@@ -47,6 +57,16 @@ export interface PromptContext extends PromptComposeParams {
|
|
|
47
57
|
add: (block: PromptBlock, text: string) => void
|
|
48
58
|
/** Resolve skill aliases through the registry, following `requires`. */
|
|
49
59
|
resolve: (aliases: readonly string[]) => SkillDefinition[]
|
|
60
|
+
/**
|
|
61
|
+
* Take exclusive ownership of `key` for THIS composition: the first caller gets `true`,
|
|
62
|
+
* every later one `false`. Two plugins that can each render the same skill — a static
|
|
63
|
+
* catalogue and a detector — would otherwise emit it twice, which costs tokens and
|
|
64
|
+
* tells the model the same thing in two voices.
|
|
65
|
+
*
|
|
66
|
+
* The claim set is per `compose` call and consulted by nobody else, so a composition
|
|
67
|
+
* where no plugin claims renders exactly the bytes it rendered before this seam existed.
|
|
68
|
+
*/
|
|
69
|
+
claim: (key: string) => boolean
|
|
50
70
|
}
|
|
51
71
|
|
|
52
72
|
/**
|
package/src/service.ts
CHANGED
|
@@ -38,11 +38,39 @@ export const llmServiceApi = (options: LlmServiceOptions, self: () => LlmService
|
|
|
38
38
|
})
|
|
39
39
|
}
|
|
40
40
|
|
|
41
|
+
/** A named alias's config, ready to be layered under something else. */
|
|
42
|
+
const presetOf = (
|
|
43
|
+
models: ModelConfig[], name: string | undefined
|
|
44
|
+
): Partial<ModelConfig> => {
|
|
45
|
+
if (name == null) return {}
|
|
46
|
+
const { alias: _alias, ...rest } = { ...(models.find(m => m.alias === name) ?? {}) }
|
|
47
|
+
return rest
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
/**
|
|
51
|
+
* Drop keys that are present but `undefined` — they must not shadow a layer below.
|
|
52
|
+
* `mergeOverride` in the execution layer strips these already; a hand-built override
|
|
53
|
+
* need not.
|
|
54
|
+
*/
|
|
55
|
+
const defined = (config: Partial<ModelConfig>): Partial<ModelConfig> =>
|
|
56
|
+
Object.fromEntries(
|
|
57
|
+
Object.entries(config).filter(([, value]) => value !== undefined)
|
|
58
|
+
) as Partial<ModelConfig>
|
|
59
|
+
|
|
41
60
|
/**
|
|
42
61
|
* Resolve `alias` → config (inheriting a `preset`, applying `override`) and build it.
|
|
43
62
|
* A declared `fallback` is built as well and attached to the primary as a
|
|
44
63
|
* non-enumerable `__fallbackModel`, which the model's retry escalator reads. The
|
|
45
64
|
* fallback spec is merged OVER the primary config, so it inherits secret/headers.
|
|
65
|
+
*
|
|
66
|
+
* Four layers, lowest first — the alias's own preset, the alias, the override's preset,
|
|
67
|
+
* the override. A `preset` is a BASE that its referent refines, so it has to sit under
|
|
68
|
+
* the config naming it; the previous order assigned it last, which meant a role
|
|
69
|
+
* declaring `preset:` silently discarded both its own fields and the caller's override —
|
|
70
|
+
* effort-tier token caps and `temperatureFactory`'s temperature among them. An override
|
|
71
|
+
* naming a preset (how the execution layer delivers a `modelOverrides` string pin) still
|
|
72
|
+
* outranks the alias, because picking a different model is a stronger statement than the
|
|
73
|
+
* role's default; explicit override fields stay on top of everything.
|
|
46
74
|
*/
|
|
47
75
|
const createModel = (alias: string, override: Partial<ModelConfig> = {}): BaseChatModel => {
|
|
48
76
|
const models = options.models()
|
|
@@ -50,17 +78,35 @@ export const llmServiceApi = (options: LlmServiceOptions, self: () => LlmService
|
|
|
50
78
|
if (baseConfig == null) {
|
|
51
79
|
throw new LlmMissconfiguredError(alias)
|
|
52
80
|
}
|
|
53
|
-
const config: ModelConfig = {
|
|
81
|
+
const config: ModelConfig = {
|
|
82
|
+
...presetOf(models, baseConfig.preset),
|
|
83
|
+
...defined(baseConfig),
|
|
84
|
+
...presetOf(models, override.preset),
|
|
85
|
+
...defined(override),
|
|
86
|
+
alias: baseConfig.alias,
|
|
87
|
+
}
|
|
54
88
|
// The service-wide idle deadline is a floor, not an override: a preset that states its
|
|
55
89
|
// own `streamTimeout` knows something specific about that model and keeps it.
|
|
56
90
|
config.streamTimeout ??= options.streamTimeout
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
91
|
+
|
|
92
|
+
// What the provider accepts bounds what we may ask for. A preset that over-declares is
|
|
93
|
+
// corrected here rather than at the provider, where it surfaces as a fatal 400 on the
|
|
94
|
+
// one call that finally escalated far enough to exceed the limit.
|
|
95
|
+
if (config.maxOutput != null && config.maxOutput > 0) {
|
|
96
|
+
if (config.maxTokensCap != null && config.maxTokensCap > config.maxOutput) {
|
|
97
|
+
console.warn(
|
|
98
|
+
`Model "${alias}" declares maxTokensCap ${config.maxTokensCap} above the provider's`
|
|
99
|
+
+ ` maxOutput ${config.maxOutput}; the escalator will stop at ${config.maxOutput}.`
|
|
100
|
+
)
|
|
101
|
+
}
|
|
102
|
+
if (config.maxTokens != null && config.maxTokens > config.maxOutput) {
|
|
103
|
+
console.warn(
|
|
104
|
+
`Model "${alias}" declares maxTokens ${config.maxTokens} above the provider's`
|
|
105
|
+
+ ` maxOutput ${config.maxOutput}; clamping.`
|
|
106
|
+
)
|
|
107
|
+
config.maxTokens = config.maxOutput
|
|
108
|
+
}
|
|
62
109
|
}
|
|
63
|
-
Object.assign(config, preset)
|
|
64
110
|
|
|
65
111
|
const { fallback, ...primaryConfig } = config
|
|
66
112
|
const primary = buildModel(alias, primaryConfig)
|