@owlmeans/llm 0.1.14

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (147) hide show
  1. package/README.md +193 -0
  2. package/agent-meta/instructions/llm.instructions.md +66 -0
  3. package/agent-meta/manifest.json +23 -0
  4. package/agent-meta/skills/llm/SKILL.md +121 -0
  5. package/build/consts.d.ts +67 -0
  6. package/build/consts.d.ts.map +1 -0
  7. package/build/consts.js +76 -0
  8. package/build/consts.js.map +1 -0
  9. package/build/errors.d.ts +33 -0
  10. package/build/errors.d.ts.map +1 -0
  11. package/build/errors.js +52 -0
  12. package/build/errors.js.map +1 -0
  13. package/build/execution/index.d.ts +4 -0
  14. package/build/execution/index.d.ts.map +1 -0
  15. package/build/execution/index.js +3 -0
  16. package/build/execution/index.js.map +1 -0
  17. package/build/execution/service.d.ts +21 -0
  18. package/build/execution/service.d.ts.map +1 -0
  19. package/build/execution/service.js +129 -0
  20. package/build/execution/service.js.map +1 -0
  21. package/build/execution/types.d.ts +119 -0
  22. package/build/execution/types.d.ts.map +1 -0
  23. package/build/execution/types.js +2 -0
  24. package/build/execution/types.js.map +1 -0
  25. package/build/execution/utils.d.ts +28 -0
  26. package/build/execution/utils.d.ts.map +1 -0
  27. package/build/execution/utils.js +60 -0
  28. package/build/execution/utils.js.map +1 -0
  29. package/build/helpers/index.d.ts +5 -0
  30. package/build/helpers/index.d.ts.map +1 -0
  31. package/build/helpers/index.js +5 -0
  32. package/build/helpers/index.js.map +1 -0
  33. package/build/helpers/json.d.ts +24 -0
  34. package/build/helpers/json.d.ts.map +1 -0
  35. package/build/helpers/json.js +119 -0
  36. package/build/helpers/json.js.map +1 -0
  37. package/build/helpers/messages.d.ts +10 -0
  38. package/build/helpers/messages.d.ts.map +1 -0
  39. package/build/helpers/messages.js +9 -0
  40. package/build/helpers/messages.js.map +1 -0
  41. package/build/helpers/retry.d.ts +18 -0
  42. package/build/helpers/retry.d.ts.map +1 -0
  43. package/build/helpers/retry.js +57 -0
  44. package/build/helpers/retry.js.map +1 -0
  45. package/build/helpers/spectate.d.ts +8 -0
  46. package/build/helpers/spectate.d.ts.map +1 -0
  47. package/build/helpers/spectate.js +54 -0
  48. package/build/helpers/spectate.js.map +1 -0
  49. package/build/index.d.ts +13 -0
  50. package/build/index.d.ts.map +1 -0
  51. package/build/index.js +11 -0
  52. package/build/index.js.map +1 -0
  53. package/build/model.d.ts +12 -0
  54. package/build/model.d.ts.map +1 -0
  55. package/build/model.js +297 -0
  56. package/build/model.js.map +1 -0
  57. package/build/plugins/anthropic.d.ts +4 -0
  58. package/build/plugins/anthropic.d.ts.map +1 -0
  59. package/build/plugins/anthropic.js +86 -0
  60. package/build/plugins/anthropic.js.map +1 -0
  61. package/build/plugins/compatible.d.ts +18 -0
  62. package/build/plugins/compatible.d.ts.map +1 -0
  63. package/build/plugins/compatible.js +54 -0
  64. package/build/plugins/compatible.js.map +1 -0
  65. package/build/plugins/export.d.ts +6 -0
  66. package/build/plugins/export.d.ts.map +1 -0
  67. package/build/plugins/export.js +5 -0
  68. package/build/plugins/export.js.map +1 -0
  69. package/build/plugins/index.d.ts +18 -0
  70. package/build/plugins/index.d.ts.map +1 -0
  71. package/build/plugins/index.js +42 -0
  72. package/build/plugins/index.js.map +1 -0
  73. package/build/plugins/openai.d.ts +28 -0
  74. package/build/plugins/openai.d.ts.map +1 -0
  75. package/build/plugins/openai.js +89 -0
  76. package/build/plugins/openai.js.map +1 -0
  77. package/build/plugins/types.d.ts +80 -0
  78. package/build/plugins/types.d.ts.map +1 -0
  79. package/build/plugins/types.js +2 -0
  80. package/build/plugins/types.js.map +1 -0
  81. package/build/plugins/utils.d.ts +27 -0
  82. package/build/plugins/utils.d.ts.map +1 -0
  83. package/build/plugins/utils.js +33 -0
  84. package/build/plugins/utils.js.map +1 -0
  85. package/build/service.d.ts +24 -0
  86. package/build/service.d.ts.map +1 -0
  87. package/build/service.js +95 -0
  88. package/build/service.js.map +1 -0
  89. package/build/types.d.ts +190 -0
  90. package/build/types.d.ts.map +1 -0
  91. package/build/types.js +2 -0
  92. package/build/types.js.map +1 -0
  93. package/build/utils/config.d.ts +13 -0
  94. package/build/utils/config.d.ts.map +1 -0
  95. package/build/utils/config.js +15 -0
  96. package/build/utils/config.js.map +1 -0
  97. package/build/utils/null-report.d.ts +36 -0
  98. package/build/utils/null-report.d.ts.map +1 -0
  99. package/build/utils/null-report.js +84 -0
  100. package/build/utils/null-report.js.map +1 -0
  101. package/build/utils/prompt.d.ts +15 -0
  102. package/build/utils/prompt.d.ts.map +1 -0
  103. package/build/utils/prompt.js +45 -0
  104. package/build/utils/prompt.js.map +1 -0
  105. package/build/utils/schema.d.ts +20 -0
  106. package/build/utils/schema.d.ts.map +1 -0
  107. package/build/utils/schema.js +28 -0
  108. package/build/utils/schema.js.map +1 -0
  109. package/build/utils/stream.d.ts +22 -0
  110. package/build/utils/stream.d.ts.map +1 -0
  111. package/build/utils/stream.js +54 -0
  112. package/build/utils/stream.js.map +1 -0
  113. package/package.json +65 -0
  114. package/src/consts.ts +89 -0
  115. package/src/errors.ts +65 -0
  116. package/src/execution/index.ts +4 -0
  117. package/src/execution/service.ts +185 -0
  118. package/src/execution/types.ts +139 -0
  119. package/src/execution/utils.ts +79 -0
  120. package/src/helpers/index.ts +5 -0
  121. package/src/helpers/json.ts +117 -0
  122. package/src/helpers/messages.ts +12 -0
  123. package/src/helpers/retry.ts +59 -0
  124. package/src/helpers/spectate.ts +67 -0
  125. package/src/index.ts +13 -0
  126. package/src/model.ts +379 -0
  127. package/src/plugins/anthropic.ts +97 -0
  128. package/src/plugins/compatible.ts +62 -0
  129. package/src/plugins/export.ts +6 -0
  130. package/src/plugins/index.ts +53 -0
  131. package/src/plugins/openai.ts +108 -0
  132. package/src/plugins/types.ts +92 -0
  133. package/src/plugins/utils.ts +38 -0
  134. package/src/service.ts +125 -0
  135. package/src/types.ts +214 -0
  136. package/src/utils/config.ts +19 -0
  137. package/src/utils/null-report.ts +126 -0
  138. package/src/utils/prompt.ts +46 -0
  139. package/src/utils/schema.ts +35 -0
  140. package/src/utils/stream.ts +58 -0
  141. package/tests/context.ts +110 -0
  142. package/tests/execution.spec.ts +200 -0
  143. package/tests/helpers.spec.ts +192 -0
  144. package/tests/internals.spec.ts +141 -0
  145. package/tests/model.spec.ts +116 -0
  146. package/tests/plugins.spec.ts +227 -0
  147. package/tsconfig.json +19 -0
@@ -0,0 +1,117 @@
1
+ /** Property names checked when a model wraps a scalar string in an object. */
2
+ const SCALAR_KEYS = ['path', 'file', 'filename', 'name', 'value', 'source']
3
+
4
+ const tryParse = (text: string): unknown => {
5
+ try {
6
+ return JSON.parse(text)
7
+ } catch {
8
+ return undefined
9
+ }
10
+ }
11
+
12
+ /**
13
+ * Recover a JSON value from message content. Some models ignore the tool they were
14
+ * pinned to and emit the schema-shaped JSON directly as plain message content, so the
15
+ * tool-call parse yields nothing while the content is still valid JSON. Tolerates
16
+ * markdown fences and leading/trailing prose by falling back to the outermost
17
+ * `{...}` / `[...]` span. Returns `null` when nothing parseable is found.
18
+ */
19
+ export const parseJsonContent = (content: unknown): unknown => {
20
+ let text: string
21
+ if (typeof content === 'string') {
22
+ text = content
23
+ } else if (Array.isArray(content)) {
24
+ text = content.map(part => typeof part === 'object' && part !== null && 'text' in (part as Record<string, unknown>)
25
+ ? String((part as Record<string, unknown>).text) : '').join('')
26
+ } else {
27
+ return null
28
+ }
29
+
30
+ text = text.trim()
31
+ if (text === '') return null
32
+
33
+ const fence = text.match(/```(?:json)?\s*([\s\S]*?)```/i)
34
+ if (fence != null) text = fence[1]!.trim()
35
+
36
+ const whole = tryParse(text)
37
+ if (whole !== undefined) return whole
38
+
39
+ const firstObj = text.indexOf('{')
40
+ const lastObj = text.lastIndexOf('}')
41
+ if (firstObj >= 0 && lastObj > firstObj) {
42
+ const parsed = tryParse(text.slice(firstObj, lastObj + 1))
43
+ if (parsed !== undefined) return parsed
44
+ }
45
+
46
+ const firstArr = text.indexOf('[')
47
+ const lastArr = text.lastIndexOf(']')
48
+ if (firstArr >= 0 && lastArr > firstArr) {
49
+ const parsed = tryParse(text.slice(firstArr, lastArr + 1))
50
+ if (parsed !== undefined) return parsed
51
+ }
52
+
53
+ return null
54
+ }
55
+
56
+ /**
57
+ * Reconcile a model's answer with the schema it was given, for the two mistakes models
58
+ * make most often with tool-call arguments:
59
+ *
60
+ * 1. **Stringified structures** — an `array` field filled with the string `"[]"`, an
61
+ * `integer` field with `"7"`, a `boolean` with `"true"`. Parsed back to the declared
62
+ * type; on a parse failure the original value is kept so validation reports the real error.
63
+ * 2. **Over-wrapped scalars** — a `string` field filled with `{ path: "…" }` instead of
64
+ * `"…"`. Unwrapped via the likely key, or via the single string property if there is
65
+ * exactly one.
66
+ *
67
+ * Walks objects and arrays, so nested occurrences are fixed too. Purely defensive: a
68
+ * value that already matches its schema is returned untouched.
69
+ */
70
+ export const coerceToSchema = (value: unknown, schema: unknown): unknown => {
71
+ if (schema == null || typeof schema !== 'object') return value
72
+ const s = schema as { type?: string; properties?: Record<string, unknown>; items?: unknown }
73
+ const type = s.type
74
+
75
+ if (typeof value === 'string' && type != null && type !== 'string') {
76
+ if (type === 'array' || type === 'object') {
77
+ try {
78
+ return coerceToSchema(JSON.parse(value), schema)
79
+ } catch {
80
+ return value
81
+ }
82
+ }
83
+ if (type === 'integer' || type === 'number') {
84
+ const n = Number(value)
85
+ return Number.isNaN(n) ? value : n
86
+ }
87
+ if (type === 'boolean') {
88
+ if (value === 'true') return true
89
+ if (value === 'false') return false
90
+ return value
91
+ }
92
+ }
93
+
94
+ if (type === 'string' && value != null && typeof value === 'object' && !Array.isArray(value)) {
95
+ const obj = value as Record<string, unknown>
96
+ for (const key of SCALAR_KEYS) {
97
+ if (typeof obj[key] === 'string') return obj[key]
98
+ }
99
+ const strings = Object.values(obj).filter((v): v is string => typeof v === 'string')
100
+ if (strings.length === 1) return strings[0]
101
+ }
102
+
103
+ if (type === 'object' && s.properties != null && value != null
104
+ && typeof value === 'object' && !Array.isArray(value)) {
105
+ const obj = value as Record<string, unknown>
106
+ for (const [key, propSchema] of Object.entries(s.properties)) {
107
+ if (key in obj) obj[key] = coerceToSchema(obj[key], propSchema)
108
+ }
109
+ return obj
110
+ }
111
+
112
+ if (type === 'array' && Array.isArray(value) && s.items != null) {
113
+ return value.map(item => coerceToSchema(item, s.items))
114
+ }
115
+
116
+ return value
117
+ }
@@ -0,0 +1,12 @@
1
+ import type { MessageFieldWithRole } from '@langchain/core/messages'
2
+ import type { ModelInput } from '../types.js'
3
+
4
+ /**
5
+ * Normalize whatever a caller passed as input into an array of role-tagged messages:
6
+ * a bare string becomes a `user` message, a single message becomes a one-element array.
7
+ * The result is a fresh array the model is free to mutate (JSON directive, `/no_think`,
8
+ * cache markers) without touching the caller's data.
9
+ */
10
+ export const normalizeInput = (input: ModelInput): MessageFieldWithRole[] =>
11
+ (Array.isArray(input) ? input : [input])
12
+ .map(message => typeof message === 'string' ? { role: 'user' as const, content: message } : message) as MessageFieldWithRole[]
@@ -0,0 +1,59 @@
1
+ import { LlmRetryExceededError } from '../errors.js'
2
+ import { plugins } from '../plugins/index.js'
3
+ import type { FatalErrorResolver, RetryOptions } from '../types.js'
4
+
5
+ const resolvers: FatalErrorResolver[] = []
6
+
7
+ /**
8
+ * Register a globally-applicable rule that turns a thrown error into an immediate
9
+ * abort of every retry loop in this package. Use it for conditions no amount of
10
+ * retrying can fix — an exhausted budget, a revoked credential, a cancelled job.
11
+ *
12
+ * Provider plugins contribute their own through `LlmPlugin.isFatal`; both sets are
13
+ * consulted, plus the per-call {@link RetryOptions.fatal}.
14
+ */
15
+ export const registerFatalError = (resolver: FatalErrorResolver): void => {
16
+ resolvers.push(resolver)
17
+ }
18
+
19
+ const resolveFatal = (e: unknown, fatal?: FatalErrorResolver): Error | null => {
20
+ const own = fatal?.(e)
21
+ if (own != null) return own
22
+ for (const resolver of resolvers) {
23
+ const found = resolver(e)
24
+ if (found != null) return found
25
+ }
26
+ for (const plugin of Object.values(plugins)) {
27
+ const found = plugin.isFatal?.(e)
28
+ if (found != null) return found
29
+ }
30
+ return null
31
+ }
32
+
33
+ /**
34
+ * Run `fn` up to `retries` times, passing the 0-based attempt number so the callee can
35
+ * escalate (a bigger output budget, a stronger model). Every non-fatal error is
36
+ * swallowed and retained as the `cause` of the {@link LlmRetryExceededError} thrown when
37
+ * the attempts run out; a fatal error (see {@link registerFatalError}) is rethrown at once.
38
+ */
39
+ export const withRetry = async <T>(
40
+ { retries, outputErrors = false, fatal }: RetryOptions,
41
+ fn: (attempt: number) => Promise<T>
42
+ ): Promise<T> => {
43
+ const exceeded = new LlmRetryExceededError('max-retries')
44
+ for (let i = 0; i < retries; ++i) {
45
+ try {
46
+ return await fn(i)
47
+ } catch (e) {
48
+ const abort = resolveFatal(e, fatal)
49
+ if (abort != null) throw abort
50
+ exceeded.cause = e
51
+ exceeded.attempt = i
52
+ if (outputErrors) {
53
+ console.debug('Retry error on attempt', i)
54
+ console.error(e)
55
+ }
56
+ }
57
+ }
58
+ throw exceeded
59
+ }
@@ -0,0 +1,67 @@
1
+ import { AIMessage, BaseMessage } from '@langchain/core/messages'
2
+ import type { UsageMetadata } from '@langchain/core/messages'
3
+ import { SpectatorContentType } from '@owlmeans/llm-common'
4
+ import type { SpectatorEntryMessage } from '@owlmeans/llm-common'
5
+ import type { LlmSpectator, ModelInputItem } from '../types.js'
6
+
7
+ /** Normalize one prompt message into the spectator's storage shape. */
8
+ const describeInput = (msg: ModelInputItem, callType: string): SpectatorEntryMessage => {
9
+ const entry: SpectatorEntryMessage = {
10
+ callType,
11
+ type: 'unknown',
12
+ content: '',
13
+ contentType: SpectatorContentType.Text,
14
+ }
15
+
16
+ if (msg instanceof BaseMessage) {
17
+ entry.type = msg.type
18
+ entry.content = msg.content as unknown as string
19
+ entry.name = msg.name
20
+ entry.contentType = typeof msg.content === 'string' ? SpectatorContentType.Text : SpectatorContentType.Json
21
+ entry.usage = 'usage_metadata' in msg ? msg.usage_metadata as UsageMetadata : undefined
22
+ entry.raw = JSON.stringify({ message: msg, content: msg.content, meta: msg.response_metadata })
23
+ } else if (typeof msg === 'object' && 'content' in msg) {
24
+ entry.type = msg.role as string ?? 'unknown'
25
+ entry.content = msg.content as string ?? msg as unknown as string
26
+ entry.raw = JSON.stringify(msg)
27
+ } else {
28
+ entry.content = msg as unknown as string
29
+ entry.raw = JSON.stringify(msg)
30
+ }
31
+
32
+ return entry
33
+ }
34
+
35
+ /**
36
+ * Log a completed model call: every prompt message plus the completion, as one entry.
37
+ * A tool-calling completion carries its arguments rather than its (empty) content.
38
+ */
39
+ export const spectate = (spectator: LlmSpectator, callType: string) =>
40
+ async (
41
+ input: ModelInputItem[],
42
+ message: AIMessage,
43
+ action: string,
44
+ retries: number,
45
+ startedAt?: number,
46
+ ) => {
47
+ const messages = input.map(msg => describeInput(msg, callType))
48
+
49
+ const completion: SpectatorEntryMessage = {
50
+ type: message.type,
51
+ callType,
52
+ content: '',
53
+ name: message.name,
54
+ contentType: typeof message.content === 'string' ? SpectatorContentType.Text : SpectatorContentType.Json,
55
+ usage: message.usage_metadata,
56
+ raw: JSON.stringify({ message, content: message.content, meta: message.response_metadata }),
57
+ }
58
+
59
+ if (message.tool_calls != null && message.tool_calls.length > 0) {
60
+ completion.content = message.tool_calls as unknown as string
61
+ completion.contentType = SpectatorContentType.ToolCall
62
+ } else {
63
+ completion.content = message.content as unknown as string
64
+ }
65
+
66
+ return spectator.log({ action, retries, startedAt, messages: [...messages, completion] })
67
+ }
package/src/index.ts ADDED
@@ -0,0 +1,13 @@
1
+
2
+ export * from './consts.js'
3
+ export * from './errors.js'
4
+ export type * from './types.js'
5
+ export * from './model.js'
6
+ export * from './service.js'
7
+ export * from './helpers/index.js'
8
+ export * from './execution/index.js'
9
+ export type * from './plugins/types.js'
10
+ export { plugins, registerLlmPlugin, pluginOf, pluginFor, resolvePlugin } from './plugins/index.js'
11
+ export { anthropicPlugin, ANTHROPIC_FAMILY } from './plugins/anthropic.js'
12
+ export { compatiblePlugin } from './plugins/compatible.js'
13
+ export { openAiPlugin, openAiFamily, OPENAI_FAMILY } from './plugins/openai.js'
package/src/model.ts ADDED
@@ -0,0 +1,379 @@
1
+ import { Ajv } from 'ajv'
2
+ import type { JSONSchemaType } from 'ajv'
3
+ import { AIMessage } from '@langchain/core/messages'
4
+ import type { AIMessageChunk, MessageFieldWithRole } from '@langchain/core/messages'
5
+ import type { BaseChatModel } from '@langchain/core/language_models/chat_models'
6
+ import { StructuredMode } from '@owlmeans/llm-common'
7
+ import type { NullKind } from '@owlmeans/llm-common'
8
+ import {
9
+ DEFAULT_MAX_OUTPUT_CAP, DEFAULT_MODEL_RETRIES, FALLBACK_AFTER_ATTEMPTS,
10
+ } from './consts.js'
11
+ import { LlmModelError } from './errors.js'
12
+ import { pluginFor, pluginOf } from './plugins/index.js'
13
+ import type { LlmPlugin } from './plugins/types.js'
14
+ import { coerceToSchema, parseJsonContent } from './helpers/json.js'
15
+ import { normalizeInput } from './helpers/messages.js'
16
+ import { withRetry } from './helpers/retry.js'
17
+ import { spectate } from './helpers/spectate.js'
18
+ import { idleTimeout, readConfig } from './utils/config.js'
19
+ import { reportNull } from './utils/null-report.js'
20
+ import type { NullReportParams } from './utils/null-report.js'
21
+ import { applyNoThink, ensureJsonMention } from './utils/prompt.js'
22
+ import { resolveSchemaValidator, toToolName, unwrapNamed } from './utils/schema.js'
23
+ import { streamWithDeadline } from './utils/stream.js'
24
+ import type {
25
+ LlmAskOptions, LlmInvokeOptions, LlmModel, LlmModelOptions, LlmRequestOptions,
26
+ LlmSpectator, LlmTalkOptions, ModelInput, RefferedResult,
27
+ } from './types.js'
28
+
29
+ type StreamOptions = Parameters<BaseChatModel['stream']>[1]
30
+
31
+ /**
32
+ * Build the four-method model API on top of a LangChain chat model.
33
+ *
34
+ * Everything provider-specific — how the client is refined between retries, how
35
+ * structured output is requested, whether prompt caching exists — is delegated to the
36
+ * `LlmPlugin` resolved for this model (see `plugins/`). The model itself only owns the
37
+ * provider-independent parts: streaming under an idle deadline, retry/fallback
38
+ * escalation, schema validation and coercion, spectator logging, and null diagnostics.
39
+ */
40
+ export const makeLlmModel = ({
41
+ model,
42
+ outputErrors = false,
43
+ captureNull = false,
44
+ retries = DEFAULT_MODEL_RETRIES,
45
+ purpose,
46
+ }: LlmModelOptions, spectator: LlmSpectator): LlmModel => {
47
+
48
+ const ajv = new Ajv({ strict: false })
49
+
50
+ // The original config and its plugin are static per model instance: a REFINED instance
51
+ // is rebuilt from `lc_kwargs` and does not reliably carry the metadata back.
52
+ const config = readConfig(model)
53
+ const plugin: LlmPlugin | undefined = pluginOf(config.provider) ?? pluginFor(model)
54
+ const timeout = idleTimeout(config)
55
+
56
+ /** Normalize, then apply every in-place prompt adaptation, in dependency order. */
57
+ const prepare = (input: ModelInput, useCache: boolean, cacheMax: number, json: boolean): MessageFieldWithRole[] => {
58
+ const msgs = normalizeInput(input)
59
+ if (json) ensureJsonMention(msgs)
60
+ applyNoThink(msgs, config.disableThinking)
61
+ // Cache markers replace string content with content blocks, so they must go last.
62
+ if (plugin?.patchCache?.(msgs, { model, useCache, cacheMax }) === true) {
63
+ console.log(`Prompt caching enabled for ${plugin.type} (up to ${cacheMax} breakpoints)`)
64
+ }
65
+ return msgs
66
+ }
67
+
68
+ const notifyRef = <T>(ref: RefferedResult<T> | undefined, value: T): void => {
69
+ if (ref != null) {
70
+ ref.value = value
71
+ void ref.callback?.(value).finally()
72
+ }
73
+ }
74
+
75
+ /**
76
+ * Record the diagnostics of a call that produced nothing usable and build the
77
+ * retryable error describing it. The caller throws it, so control flow stays visible.
78
+ */
79
+ const nullResult = async (
80
+ kind: NullKind,
81
+ p: Omit<NullReportParams, 'kind' | 'purpose' | 'config'>,
82
+ parsed: boolean = false,
83
+ ): Promise<LlmModelError> => {
84
+ await reportNull(spectator, captureNull, { ...p, kind, purpose, config })
85
+ const toolCalls = (p.raw as unknown as { tool_calls?: unknown[] } | null)?.tool_calls?.length ?? 0
86
+ const content = JSON.stringify(p.raw?.content ?? null).substring(0, 120)
87
+ return new LlmModelError(
88
+ `null-output:raw=${p.raw != null}, parsed=${parsed}, toolCalls=${toolCalls}, content=${content}`
89
+ )
90
+ }
91
+
92
+ /**
93
+ * Rebuild the model for attempt N.
94
+ *
95
+ * Two-layer fallback: once a cheap primary has failed {@link FALLBACK_AFTER_ATTEMPTS}
96
+ * times, escalate to the stronger model the service attached as `__fallbackModel`. The
97
+ * escalation only happens WITHIN one plugin family — rotating providers mid-call would
98
+ * flip the structured-output call shape (tool_choice spelling, native support), so it is
99
+ * better to keep retrying on the primary than to switch families.
100
+ */
101
+ const refineModel = (attempt: number, temperature?: number): BaseChatModel => {
102
+ const fallbackModel = (model as unknown as { __fallbackModel?: BaseChatModel }).__fallbackModel
103
+ const sameFamily = fallbackModel != null && plugin != null
104
+ && plugin.owns(model) && plugin.owns(fallbackModel)
105
+ const base = (sameFamily && attempt >= FALLBACK_AFTER_ATTEMPTS) ? fallbackModel : model
106
+
107
+ if (attempt === FALLBACK_AFTER_ATTEMPTS && fallbackModel != null) {
108
+ if (base !== model) {
109
+ console.warn(`${base.getName()}: switching to fallback model after ${attempt} failed attempts`)
110
+ } else {
111
+ console.warn(
112
+ `Skipping cross-family fallback (${fallbackModel.getName()}); staying on ${model.getName()}`
113
+ )
114
+ }
115
+ }
116
+
117
+ const baseConfig = readConfig(base)
118
+ const basePlugin = pluginOf(baseConfig.provider) ?? pluginFor(base)
119
+ if (basePlugin == null) return base
120
+
121
+ const maxOutputCap = typeof baseConfig.maxTokensCap === 'number' && baseConfig.maxTokensCap > 0
122
+ ? baseConfig.maxTokensCap
123
+ : DEFAULT_MAX_OUTPUT_CAP
124
+ const refined = basePlugin.refine({ base, attempt, temperature, maxOutputCap })
125
+
126
+ if (attempt > 0) {
127
+ const maxTokens = (refined as unknown as { maxTokens?: number }).maxTokens
128
+ console.warn(`${refined.getName()}: retry attempt ${attempt}, maxTokens now ${maxTokens}`)
129
+ }
130
+
131
+ return refined
132
+ }
133
+
134
+ /** How this model should be asked for schema-conforming output. */
135
+ const structuredMode = (): StructuredMode => plugin != null
136
+ ? plugin.structuredMode(config as Parameters<LlmPlugin['structuredMode']>[0])
137
+ : (config.structuredOutput === true ? StructuredMode.Native : StructuredMode.Tool)
138
+
139
+ /**
140
+ * Shared structured-output core for `invoke`/`request`. Streams under the idle deadline
141
+ * (whose break-on-`finish_reason` also dedups the duplicate final chunk some providers
142
+ * emit, which would otherwise corrupt accumulated tool-call arguments), accumulates the
143
+ * chunks and extracts the parsed object.
144
+ *
145
+ * `withStructuredOutput` is deliberately NOT used: it buffers the whole raw stream
146
+ * before yielding, which is too late to dedup.
147
+ */
148
+ const streamStructured = async <T>(
149
+ refined: BaseChatModel,
150
+ msgs: MessageFieldWithRole[],
151
+ innerSchema: JSONSchemaType<T>,
152
+ toolName: string,
153
+ action: string,
154
+ ): Promise<{ piece: AIMessageChunk | null; result: T | null; mode: StructuredMode }> => {
155
+ const responseFormat = structuredMode() === StructuredMode.Native
156
+ ? plugin?.responseFormat?.(toolName, innerSchema)
157
+ : undefined
158
+ // A plugin that declares Native but provides no response_format falls back to tools.
159
+ const mode = responseFormat != null ? StructuredMode.Native : StructuredMode.Tool
160
+
161
+ const start = (signal: AbortSignal): Promise<AsyncIterable<unknown>> => {
162
+ const base = { runName: action, metadata: { purpose }, signal }
163
+ if (mode === StructuredMode.Native) {
164
+ return refined.stream(msgs, { ...base, ...responseFormat && { response_format: responseFormat } } as unknown as StreamOptions)
165
+ }
166
+ // langchain converts the OpenAI-shaped tool DEFINITION for either provider, but the
167
+ // `tool_choice` shape is NOT converted — the plugin supplies the right spelling.
168
+ // eslint-disable-next-line @typescript-eslint/no-non-null-assertion
169
+ const bound = refined.bindTools!(
170
+ [{ type: 'function', function: { name: toolName, description: '', parameters: innerSchema } }],
171
+ { tool_choice: plugin?.toolChoice(toolName) ?? { type: 'function', function: { name: toolName } } }
172
+ )
173
+ return bound.stream(msgs, base)
174
+ }
175
+
176
+ let piece: AIMessageChunk | null = null
177
+ for await (const rawChunk of streamWithDeadline(start, timeout)) {
178
+ const chunk = rawChunk as AIMessageChunk
179
+ piece = piece == null ? chunk : (piece.concat(chunk) as AIMessageChunk)
180
+ }
181
+
182
+ let result: T | null = null
183
+ if (mode === StructuredMode.Tool) {
184
+ const rawArgs = piece?.tool_calls?.[0]?.args ?? null
185
+ result = rawArgs != null
186
+ ? (typeof rawArgs === 'string' ? parseJsonContent(rawArgs) as T | null : rawArgs as T)
187
+ : null
188
+ }
189
+ // Content fallback: some models ignore the pinned tool (or the JSON mode) and emit the
190
+ // schema-shaped JSON as plain content.
191
+ if (result == null && piece != null) {
192
+ const recovered = parseJsonContent(piece.content)
193
+ if (recovered != null) result = recovered as T
194
+ }
195
+
196
+ return { piece, result, mode }
197
+ }
198
+
199
+ const helper: LlmModel = {
200
+ ask: async (input, { ref, filter, action, useCache = false, cacheMax = 4 }: LlmAskOptions) => {
201
+ const msgs = prepare(input, useCache, cacheMax, false)
202
+ return withRetry({ retries, outputErrors }, async i => {
203
+ const refined = refineModel(i)
204
+ console.log('Use model to ask: ', refined.getName(), refined.lc_kwargs.model)
205
+ const startedAt = Date.now()
206
+ let result: AIMessageChunk | null = null
207
+ for await (const chunk of streamWithDeadline(
208
+ signal => refined.stream(msgs, { runName: action, metadata: { purpose }, signal }), timeout
209
+ )) {
210
+ result = result == null ? chunk : result.concat(chunk)
211
+ }
212
+ if (result == null) {
213
+ throw await nullResult('ask', { action, attempt: i, startedAt, refined, msgs, raw: result, useCache })
214
+ }
215
+
216
+ const message = new AIMessage(result)
217
+ let output: string | null = typeof result.content === 'string'
218
+ ? result.content
219
+ : Array.isArray(result.content)
220
+ ? result.content
221
+ .filter((c): c is { type: string; text: string } =>
222
+ typeof c === 'object' && c !== null && 'type' in c && c.type === 'text'
223
+ )
224
+ .map(c => c.text)
225
+ .join('') || null
226
+ : null
227
+
228
+ const entry = await spectate(spectator, 'ask')(msgs, message, action, i, startedAt)
229
+ if (ref != null) ref.spectatorEntry = entry
230
+
231
+ if (filter != null) {
232
+ output = await filter(output ?? '', message)
233
+ if (output == null) {
234
+ throw new LlmModelError(`filter-rejected:${JSON.stringify(message).substring(0, 50)}...`)
235
+ }
236
+ } else if (output == null || output.trim() === '') {
237
+ throw new LlmModelError(`empty-content:${JSON.stringify(message).substring(0, 50)}...`)
238
+ }
239
+
240
+ notifyRef(ref, message)
241
+ return output
242
+ })
243
+ },
244
+
245
+ talk: async (input, { ref, filter, action, useCache = false, cacheMax = 4 }: LlmTalkOptions) => {
246
+ const msgs = prepare(input, useCache, cacheMax, false)
247
+ return withRetry({ retries, outputErrors }, async i => {
248
+ const refined = refineModel(i)
249
+ console.log('Use model to talk: ', refined.getName(), refined.lc_kwargs.model)
250
+ const startedAt = Date.now()
251
+ let result: AIMessageChunk | null = null
252
+ for await (const chunk of streamWithDeadline(
253
+ signal => refined.stream(msgs, { runName: action, metadata: { purpose }, signal }), timeout
254
+ )) {
255
+ result = result == null ? chunk : result.concat(chunk)
256
+ }
257
+ if (result == null) {
258
+ throw await nullResult('talk', { action, attempt: i, startedAt, refined, msgs, raw: result, useCache })
259
+ }
260
+
261
+ let message: AIMessage | null = new AIMessage(result)
262
+ const entry = await spectate(spectator, 'talk')(msgs, message, action, i, startedAt)
263
+ if (ref != null) ref.spectatorEntry = entry
264
+
265
+ if (filter != null) {
266
+ message = await filter(message)
267
+ if (message == null) {
268
+ throw new LlmModelError('filter-rejected:talk')
269
+ }
270
+ }
271
+
272
+ notifyRef(ref, message)
273
+ return message
274
+ })
275
+ },
276
+
277
+ invoke: async <T>(
278
+ input: ModelInput,
279
+ schema: JSONSchemaType<T>,
280
+ { temperature, ref, filter, action, useCache = false, cacheMax = 4 }: LlmInvokeOptions<T>
281
+ ) => {
282
+ const msgs = prepare(input, useCache, cacheMax, true)
283
+ const { name, innerSchema, validate } = resolveSchemaValidator<T>(ajv, schema)
284
+ const toolName = toToolName((innerSchema as { title?: string }).title ?? name)
285
+
286
+ return withRetry({ retries, outputErrors }, async i => {
287
+ const refined = refineModel(i, temperature)
288
+ console.log('Use model invoke: ', refined.getName(), refined.lc_kwargs.model)
289
+ const startedAt = Date.now()
290
+ const { piece, result: collected } = await streamStructured(refined, msgs, innerSchema, toolName, action)
291
+ let result: T | null = collected
292
+ if (piece == null || result == null) {
293
+ throw await nullResult('invoke', {
294
+ action, attempt: i, startedAt, refined, msgs, raw: piece,
295
+ schema: { toolName, innerSchema }, useCache,
296
+ }, result != null)
297
+ }
298
+
299
+ const message = new AIMessage(piece)
300
+ const entry = await spectate(spectator, 'invoke')(msgs, message, action, i, startedAt)
301
+ if (ref != null) ref.spectatorEntry = entry
302
+
303
+ result = unwrapNamed(result, name)
304
+ result = coerceToSchema(result, innerSchema) as T
305
+
306
+ if (filter != null) {
307
+ const preFilter = result
308
+ result = await filter(result, message)
309
+ if (result == null) {
310
+ throw new LlmModelError(`filter-rejected:${JSON.stringify(preFilter).substring(0, 200)}`)
311
+ }
312
+ }
313
+
314
+ if (!validate(result)) {
315
+ const err = new LlmModelError(`validation-failed:${JSON.stringify(validate.errors)}`)
316
+ err.cause = validate.errors?.[0]
317
+ throw err
318
+ }
319
+
320
+ notifyRef(ref, message)
321
+ return result as T
322
+ })
323
+ },
324
+
325
+ request: async <T>(
326
+ input: ModelInput,
327
+ schema: JSONSchemaType<T>,
328
+ { ref, filter, action, useCache = false, cacheMax = 4 }: LlmRequestOptions
329
+ ) => {
330
+ const msgs = prepare(input, useCache, cacheMax, true)
331
+ const { name, innerSchema, validate } = resolveSchemaValidator<T>(ajv, schema)
332
+ const toolName = toToolName((innerSchema as { title?: string }).title ?? name)
333
+
334
+ return withRetry({ retries, outputErrors }, async i => {
335
+ const refined = refineModel(i)
336
+ console.log('Use model request: ', refined.getName(), refined.lc_kwargs.model)
337
+ const startedAt = Date.now()
338
+ const { piece, result: collected } = await streamStructured(refined, msgs, innerSchema, toolName, action)
339
+ let result: T | null = collected
340
+ if (piece == null || result == null) {
341
+ throw await nullResult('request', {
342
+ action, attempt: i, startedAt, refined, msgs, raw: piece,
343
+ schema: { toolName, innerSchema }, useCache,
344
+ }, result != null)
345
+ }
346
+
347
+ let message: AIMessage | null = new AIMessage(piece)
348
+ const entry = await spectate(spectator, 'request')(msgs, message, action, i, startedAt)
349
+ if (ref != null) ref.spectatorEntry = entry
350
+
351
+ // Structured output delivers the JSON as parsed tool arguments, not as message
352
+ // content. Surface it on `.content` so the AIMessage contract callers rely on holds.
353
+ result = unwrapNamed(result, name)
354
+ result = coerceToSchema(result, innerSchema) as T
355
+ message.content = JSON.stringify(result)
356
+
357
+ if (filter != null) {
358
+ message = await filter(message)
359
+ if (message == null) {
360
+ throw new LlmModelError('filter-rejected:request')
361
+ }
362
+ } else if (message.content == null || message.content.toString().trim() === '') {
363
+ throw new LlmModelError(`empty-content:${JSON.stringify(message).substring(0, 50)}...`)
364
+ }
365
+
366
+ if (!validate(JSON.parse(`${message.content}`))) {
367
+ const err = new LlmModelError(`validation-failed:${JSON.stringify(validate.errors)}`)
368
+ err.cause = validate.errors?.[0]
369
+ throw err
370
+ }
371
+
372
+ notifyRef(ref, message)
373
+ return message
374
+ })
375
+ },
376
+ }
377
+
378
+ return helper
379
+ }