@owlmeans/llm 0.1.14
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +193 -0
- package/agent-meta/instructions/llm.instructions.md +66 -0
- package/agent-meta/manifest.json +23 -0
- package/agent-meta/skills/llm/SKILL.md +121 -0
- package/build/consts.d.ts +67 -0
- package/build/consts.d.ts.map +1 -0
- package/build/consts.js +76 -0
- package/build/consts.js.map +1 -0
- package/build/errors.d.ts +33 -0
- package/build/errors.d.ts.map +1 -0
- package/build/errors.js +52 -0
- package/build/errors.js.map +1 -0
- package/build/execution/index.d.ts +4 -0
- package/build/execution/index.d.ts.map +1 -0
- package/build/execution/index.js +3 -0
- package/build/execution/index.js.map +1 -0
- package/build/execution/service.d.ts +21 -0
- package/build/execution/service.d.ts.map +1 -0
- package/build/execution/service.js +129 -0
- package/build/execution/service.js.map +1 -0
- package/build/execution/types.d.ts +119 -0
- package/build/execution/types.d.ts.map +1 -0
- package/build/execution/types.js +2 -0
- package/build/execution/types.js.map +1 -0
- package/build/execution/utils.d.ts +28 -0
- package/build/execution/utils.d.ts.map +1 -0
- package/build/execution/utils.js +60 -0
- package/build/execution/utils.js.map +1 -0
- package/build/helpers/index.d.ts +5 -0
- package/build/helpers/index.d.ts.map +1 -0
- package/build/helpers/index.js +5 -0
- package/build/helpers/index.js.map +1 -0
- package/build/helpers/json.d.ts +24 -0
- package/build/helpers/json.d.ts.map +1 -0
- package/build/helpers/json.js +119 -0
- package/build/helpers/json.js.map +1 -0
- package/build/helpers/messages.d.ts +10 -0
- package/build/helpers/messages.d.ts.map +1 -0
- package/build/helpers/messages.js +9 -0
- package/build/helpers/messages.js.map +1 -0
- package/build/helpers/retry.d.ts +18 -0
- package/build/helpers/retry.d.ts.map +1 -0
- package/build/helpers/retry.js +57 -0
- package/build/helpers/retry.js.map +1 -0
- package/build/helpers/spectate.d.ts +8 -0
- package/build/helpers/spectate.d.ts.map +1 -0
- package/build/helpers/spectate.js +54 -0
- package/build/helpers/spectate.js.map +1 -0
- package/build/index.d.ts +13 -0
- package/build/index.d.ts.map +1 -0
- package/build/index.js +11 -0
- package/build/index.js.map +1 -0
- package/build/model.d.ts +12 -0
- package/build/model.d.ts.map +1 -0
- package/build/model.js +297 -0
- package/build/model.js.map +1 -0
- package/build/plugins/anthropic.d.ts +4 -0
- package/build/plugins/anthropic.d.ts.map +1 -0
- package/build/plugins/anthropic.js +86 -0
- package/build/plugins/anthropic.js.map +1 -0
- package/build/plugins/compatible.d.ts +18 -0
- package/build/plugins/compatible.d.ts.map +1 -0
- package/build/plugins/compatible.js +54 -0
- package/build/plugins/compatible.js.map +1 -0
- package/build/plugins/export.d.ts +6 -0
- package/build/plugins/export.d.ts.map +1 -0
- package/build/plugins/export.js +5 -0
- package/build/plugins/export.js.map +1 -0
- package/build/plugins/index.d.ts +18 -0
- package/build/plugins/index.d.ts.map +1 -0
- package/build/plugins/index.js +42 -0
- package/build/plugins/index.js.map +1 -0
- package/build/plugins/openai.d.ts +28 -0
- package/build/plugins/openai.d.ts.map +1 -0
- package/build/plugins/openai.js +89 -0
- package/build/plugins/openai.js.map +1 -0
- package/build/plugins/types.d.ts +80 -0
- package/build/plugins/types.d.ts.map +1 -0
- package/build/plugins/types.js +2 -0
- package/build/plugins/types.js.map +1 -0
- package/build/plugins/utils.d.ts +27 -0
- package/build/plugins/utils.d.ts.map +1 -0
- package/build/plugins/utils.js +33 -0
- package/build/plugins/utils.js.map +1 -0
- package/build/service.d.ts +24 -0
- package/build/service.d.ts.map +1 -0
- package/build/service.js +95 -0
- package/build/service.js.map +1 -0
- package/build/types.d.ts +190 -0
- package/build/types.d.ts.map +1 -0
- package/build/types.js +2 -0
- package/build/types.js.map +1 -0
- package/build/utils/config.d.ts +13 -0
- package/build/utils/config.d.ts.map +1 -0
- package/build/utils/config.js +15 -0
- package/build/utils/config.js.map +1 -0
- package/build/utils/null-report.d.ts +36 -0
- package/build/utils/null-report.d.ts.map +1 -0
- package/build/utils/null-report.js +84 -0
- package/build/utils/null-report.js.map +1 -0
- package/build/utils/prompt.d.ts +15 -0
- package/build/utils/prompt.d.ts.map +1 -0
- package/build/utils/prompt.js +45 -0
- package/build/utils/prompt.js.map +1 -0
- package/build/utils/schema.d.ts +20 -0
- package/build/utils/schema.d.ts.map +1 -0
- package/build/utils/schema.js +28 -0
- package/build/utils/schema.js.map +1 -0
- package/build/utils/stream.d.ts +22 -0
- package/build/utils/stream.d.ts.map +1 -0
- package/build/utils/stream.js +54 -0
- package/build/utils/stream.js.map +1 -0
- package/package.json +65 -0
- package/src/consts.ts +89 -0
- package/src/errors.ts +65 -0
- package/src/execution/index.ts +4 -0
- package/src/execution/service.ts +185 -0
- package/src/execution/types.ts +139 -0
- package/src/execution/utils.ts +79 -0
- package/src/helpers/index.ts +5 -0
- package/src/helpers/json.ts +117 -0
- package/src/helpers/messages.ts +12 -0
- package/src/helpers/retry.ts +59 -0
- package/src/helpers/spectate.ts +67 -0
- package/src/index.ts +13 -0
- package/src/model.ts +379 -0
- package/src/plugins/anthropic.ts +97 -0
- package/src/plugins/compatible.ts +62 -0
- package/src/plugins/export.ts +6 -0
- package/src/plugins/index.ts +53 -0
- package/src/plugins/openai.ts +108 -0
- package/src/plugins/types.ts +92 -0
- package/src/plugins/utils.ts +38 -0
- package/src/service.ts +125 -0
- package/src/types.ts +214 -0
- package/src/utils/config.ts +19 -0
- package/src/utils/null-report.ts +126 -0
- package/src/utils/prompt.ts +46 -0
- package/src/utils/schema.ts +35 -0
- package/src/utils/stream.ts +58 -0
- package/tests/context.ts +110 -0
- package/tests/execution.spec.ts +200 -0
- package/tests/helpers.spec.ts +192 -0
- package/tests/internals.spec.ts +141 -0
- package/tests/model.spec.ts +116 -0
- package/tests/plugins.spec.ts +227 -0
- package/tsconfig.json +19 -0
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
/** Property names checked when a model wraps a scalar string in an object. */
|
|
2
|
+
const SCALAR_KEYS = ['path', 'file', 'filename', 'name', 'value', 'source']
|
|
3
|
+
|
|
4
|
+
const tryParse = (text: string): unknown => {
|
|
5
|
+
try {
|
|
6
|
+
return JSON.parse(text)
|
|
7
|
+
} catch {
|
|
8
|
+
return undefined
|
|
9
|
+
}
|
|
10
|
+
}
|
|
11
|
+
|
|
12
|
+
/**
|
|
13
|
+
* Recover a JSON value from message content. Some models ignore the tool they were
|
|
14
|
+
* pinned to and emit the schema-shaped JSON directly as plain message content, so the
|
|
15
|
+
* tool-call parse yields nothing while the content is still valid JSON. Tolerates
|
|
16
|
+
* markdown fences and leading/trailing prose by falling back to the outermost
|
|
17
|
+
* `{...}` / `[...]` span. Returns `null` when nothing parseable is found.
|
|
18
|
+
*/
|
|
19
|
+
export const parseJsonContent = (content: unknown): unknown => {
|
|
20
|
+
let text: string
|
|
21
|
+
if (typeof content === 'string') {
|
|
22
|
+
text = content
|
|
23
|
+
} else if (Array.isArray(content)) {
|
|
24
|
+
text = content.map(part => typeof part === 'object' && part !== null && 'text' in (part as Record<string, unknown>)
|
|
25
|
+
? String((part as Record<string, unknown>).text) : '').join('')
|
|
26
|
+
} else {
|
|
27
|
+
return null
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
text = text.trim()
|
|
31
|
+
if (text === '') return null
|
|
32
|
+
|
|
33
|
+
const fence = text.match(/```(?:json)?\s*([\s\S]*?)```/i)
|
|
34
|
+
if (fence != null) text = fence[1]!.trim()
|
|
35
|
+
|
|
36
|
+
const whole = tryParse(text)
|
|
37
|
+
if (whole !== undefined) return whole
|
|
38
|
+
|
|
39
|
+
const firstObj = text.indexOf('{')
|
|
40
|
+
const lastObj = text.lastIndexOf('}')
|
|
41
|
+
if (firstObj >= 0 && lastObj > firstObj) {
|
|
42
|
+
const parsed = tryParse(text.slice(firstObj, lastObj + 1))
|
|
43
|
+
if (parsed !== undefined) return parsed
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
const firstArr = text.indexOf('[')
|
|
47
|
+
const lastArr = text.lastIndexOf(']')
|
|
48
|
+
if (firstArr >= 0 && lastArr > firstArr) {
|
|
49
|
+
const parsed = tryParse(text.slice(firstArr, lastArr + 1))
|
|
50
|
+
if (parsed !== undefined) return parsed
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
return null
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
/**
|
|
57
|
+
* Reconcile a model's answer with the schema it was given, for the two mistakes models
|
|
58
|
+
* make most often with tool-call arguments:
|
|
59
|
+
*
|
|
60
|
+
* 1. **Stringified structures** — an `array` field filled with the string `"[]"`, an
|
|
61
|
+
* `integer` field with `"7"`, a `boolean` with `"true"`. Parsed back to the declared
|
|
62
|
+
* type; on a parse failure the original value is kept so validation reports the real error.
|
|
63
|
+
* 2. **Over-wrapped scalars** — a `string` field filled with `{ path: "…" }` instead of
|
|
64
|
+
* `"…"`. Unwrapped via the likely key, or via the single string property if there is
|
|
65
|
+
* exactly one.
|
|
66
|
+
*
|
|
67
|
+
* Walks objects and arrays, so nested occurrences are fixed too. Purely defensive: a
|
|
68
|
+
* value that already matches its schema is returned untouched.
|
|
69
|
+
*/
|
|
70
|
+
export const coerceToSchema = (value: unknown, schema: unknown): unknown => {
|
|
71
|
+
if (schema == null || typeof schema !== 'object') return value
|
|
72
|
+
const s = schema as { type?: string; properties?: Record<string, unknown>; items?: unknown }
|
|
73
|
+
const type = s.type
|
|
74
|
+
|
|
75
|
+
if (typeof value === 'string' && type != null && type !== 'string') {
|
|
76
|
+
if (type === 'array' || type === 'object') {
|
|
77
|
+
try {
|
|
78
|
+
return coerceToSchema(JSON.parse(value), schema)
|
|
79
|
+
} catch {
|
|
80
|
+
return value
|
|
81
|
+
}
|
|
82
|
+
}
|
|
83
|
+
if (type === 'integer' || type === 'number') {
|
|
84
|
+
const n = Number(value)
|
|
85
|
+
return Number.isNaN(n) ? value : n
|
|
86
|
+
}
|
|
87
|
+
if (type === 'boolean') {
|
|
88
|
+
if (value === 'true') return true
|
|
89
|
+
if (value === 'false') return false
|
|
90
|
+
return value
|
|
91
|
+
}
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
if (type === 'string' && value != null && typeof value === 'object' && !Array.isArray(value)) {
|
|
95
|
+
const obj = value as Record<string, unknown>
|
|
96
|
+
for (const key of SCALAR_KEYS) {
|
|
97
|
+
if (typeof obj[key] === 'string') return obj[key]
|
|
98
|
+
}
|
|
99
|
+
const strings = Object.values(obj).filter((v): v is string => typeof v === 'string')
|
|
100
|
+
if (strings.length === 1) return strings[0]
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
if (type === 'object' && s.properties != null && value != null
|
|
104
|
+
&& typeof value === 'object' && !Array.isArray(value)) {
|
|
105
|
+
const obj = value as Record<string, unknown>
|
|
106
|
+
for (const [key, propSchema] of Object.entries(s.properties)) {
|
|
107
|
+
if (key in obj) obj[key] = coerceToSchema(obj[key], propSchema)
|
|
108
|
+
}
|
|
109
|
+
return obj
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
if (type === 'array' && Array.isArray(value) && s.items != null) {
|
|
113
|
+
return value.map(item => coerceToSchema(item, s.items))
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
return value
|
|
117
|
+
}
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
import type { MessageFieldWithRole } from '@langchain/core/messages'
|
|
2
|
+
import type { ModelInput } from '../types.js'
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* Normalize whatever a caller passed as input into an array of role-tagged messages:
|
|
6
|
+
* a bare string becomes a `user` message, a single message becomes a one-element array.
|
|
7
|
+
* The result is a fresh array the model is free to mutate (JSON directive, `/no_think`,
|
|
8
|
+
* cache markers) without touching the caller's data.
|
|
9
|
+
*/
|
|
10
|
+
export const normalizeInput = (input: ModelInput): MessageFieldWithRole[] =>
|
|
11
|
+
(Array.isArray(input) ? input : [input])
|
|
12
|
+
.map(message => typeof message === 'string' ? { role: 'user' as const, content: message } : message) as MessageFieldWithRole[]
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
import { LlmRetryExceededError } from '../errors.js'
|
|
2
|
+
import { plugins } from '../plugins/index.js'
|
|
3
|
+
import type { FatalErrorResolver, RetryOptions } from '../types.js'
|
|
4
|
+
|
|
5
|
+
const resolvers: FatalErrorResolver[] = []
|
|
6
|
+
|
|
7
|
+
/**
|
|
8
|
+
* Register a globally-applicable rule that turns a thrown error into an immediate
|
|
9
|
+
* abort of every retry loop in this package. Use it for conditions no amount of
|
|
10
|
+
* retrying can fix — an exhausted budget, a revoked credential, a cancelled job.
|
|
11
|
+
*
|
|
12
|
+
* Provider plugins contribute their own through `LlmPlugin.isFatal`; both sets are
|
|
13
|
+
* consulted, plus the per-call {@link RetryOptions.fatal}.
|
|
14
|
+
*/
|
|
15
|
+
export const registerFatalError = (resolver: FatalErrorResolver): void => {
|
|
16
|
+
resolvers.push(resolver)
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
const resolveFatal = (e: unknown, fatal?: FatalErrorResolver): Error | null => {
|
|
20
|
+
const own = fatal?.(e)
|
|
21
|
+
if (own != null) return own
|
|
22
|
+
for (const resolver of resolvers) {
|
|
23
|
+
const found = resolver(e)
|
|
24
|
+
if (found != null) return found
|
|
25
|
+
}
|
|
26
|
+
for (const plugin of Object.values(plugins)) {
|
|
27
|
+
const found = plugin.isFatal?.(e)
|
|
28
|
+
if (found != null) return found
|
|
29
|
+
}
|
|
30
|
+
return null
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
/**
|
|
34
|
+
* Run `fn` up to `retries` times, passing the 0-based attempt number so the callee can
|
|
35
|
+
* escalate (a bigger output budget, a stronger model). Every non-fatal error is
|
|
36
|
+
* swallowed and retained as the `cause` of the {@link LlmRetryExceededError} thrown when
|
|
37
|
+
* the attempts run out; a fatal error (see {@link registerFatalError}) is rethrown at once.
|
|
38
|
+
*/
|
|
39
|
+
export const withRetry = async <T>(
|
|
40
|
+
{ retries, outputErrors = false, fatal }: RetryOptions,
|
|
41
|
+
fn: (attempt: number) => Promise<T>
|
|
42
|
+
): Promise<T> => {
|
|
43
|
+
const exceeded = new LlmRetryExceededError('max-retries')
|
|
44
|
+
for (let i = 0; i < retries; ++i) {
|
|
45
|
+
try {
|
|
46
|
+
return await fn(i)
|
|
47
|
+
} catch (e) {
|
|
48
|
+
const abort = resolveFatal(e, fatal)
|
|
49
|
+
if (abort != null) throw abort
|
|
50
|
+
exceeded.cause = e
|
|
51
|
+
exceeded.attempt = i
|
|
52
|
+
if (outputErrors) {
|
|
53
|
+
console.debug('Retry error on attempt', i)
|
|
54
|
+
console.error(e)
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
}
|
|
58
|
+
throw exceeded
|
|
59
|
+
}
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
import { AIMessage, BaseMessage } from '@langchain/core/messages'
|
|
2
|
+
import type { UsageMetadata } from '@langchain/core/messages'
|
|
3
|
+
import { SpectatorContentType } from '@owlmeans/llm-common'
|
|
4
|
+
import type { SpectatorEntryMessage } from '@owlmeans/llm-common'
|
|
5
|
+
import type { LlmSpectator, ModelInputItem } from '../types.js'
|
|
6
|
+
|
|
7
|
+
/** Normalize one prompt message into the spectator's storage shape. */
|
|
8
|
+
const describeInput = (msg: ModelInputItem, callType: string): SpectatorEntryMessage => {
|
|
9
|
+
const entry: SpectatorEntryMessage = {
|
|
10
|
+
callType,
|
|
11
|
+
type: 'unknown',
|
|
12
|
+
content: '',
|
|
13
|
+
contentType: SpectatorContentType.Text,
|
|
14
|
+
}
|
|
15
|
+
|
|
16
|
+
if (msg instanceof BaseMessage) {
|
|
17
|
+
entry.type = msg.type
|
|
18
|
+
entry.content = msg.content as unknown as string
|
|
19
|
+
entry.name = msg.name
|
|
20
|
+
entry.contentType = typeof msg.content === 'string' ? SpectatorContentType.Text : SpectatorContentType.Json
|
|
21
|
+
entry.usage = 'usage_metadata' in msg ? msg.usage_metadata as UsageMetadata : undefined
|
|
22
|
+
entry.raw = JSON.stringify({ message: msg, content: msg.content, meta: msg.response_metadata })
|
|
23
|
+
} else if (typeof msg === 'object' && 'content' in msg) {
|
|
24
|
+
entry.type = msg.role as string ?? 'unknown'
|
|
25
|
+
entry.content = msg.content as string ?? msg as unknown as string
|
|
26
|
+
entry.raw = JSON.stringify(msg)
|
|
27
|
+
} else {
|
|
28
|
+
entry.content = msg as unknown as string
|
|
29
|
+
entry.raw = JSON.stringify(msg)
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
return entry
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
/**
|
|
36
|
+
* Log a completed model call: every prompt message plus the completion, as one entry.
|
|
37
|
+
* A tool-calling completion carries its arguments rather than its (empty) content.
|
|
38
|
+
*/
|
|
39
|
+
export const spectate = (spectator: LlmSpectator, callType: string) =>
|
|
40
|
+
async (
|
|
41
|
+
input: ModelInputItem[],
|
|
42
|
+
message: AIMessage,
|
|
43
|
+
action: string,
|
|
44
|
+
retries: number,
|
|
45
|
+
startedAt?: number,
|
|
46
|
+
) => {
|
|
47
|
+
const messages = input.map(msg => describeInput(msg, callType))
|
|
48
|
+
|
|
49
|
+
const completion: SpectatorEntryMessage = {
|
|
50
|
+
type: message.type,
|
|
51
|
+
callType,
|
|
52
|
+
content: '',
|
|
53
|
+
name: message.name,
|
|
54
|
+
contentType: typeof message.content === 'string' ? SpectatorContentType.Text : SpectatorContentType.Json,
|
|
55
|
+
usage: message.usage_metadata,
|
|
56
|
+
raw: JSON.stringify({ message, content: message.content, meta: message.response_metadata }),
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
if (message.tool_calls != null && message.tool_calls.length > 0) {
|
|
60
|
+
completion.content = message.tool_calls as unknown as string
|
|
61
|
+
completion.contentType = SpectatorContentType.ToolCall
|
|
62
|
+
} else {
|
|
63
|
+
completion.content = message.content as unknown as string
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
return spectator.log({ action, retries, startedAt, messages: [...messages, completion] })
|
|
67
|
+
}
|
package/src/index.ts
ADDED
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
|
|
2
|
+
export * from './consts.js'
|
|
3
|
+
export * from './errors.js'
|
|
4
|
+
export type * from './types.js'
|
|
5
|
+
export * from './model.js'
|
|
6
|
+
export * from './service.js'
|
|
7
|
+
export * from './helpers/index.js'
|
|
8
|
+
export * from './execution/index.js'
|
|
9
|
+
export type * from './plugins/types.js'
|
|
10
|
+
export { plugins, registerLlmPlugin, pluginOf, pluginFor, resolvePlugin } from './plugins/index.js'
|
|
11
|
+
export { anthropicPlugin, ANTHROPIC_FAMILY } from './plugins/anthropic.js'
|
|
12
|
+
export { compatiblePlugin } from './plugins/compatible.js'
|
|
13
|
+
export { openAiPlugin, openAiFamily, OPENAI_FAMILY } from './plugins/openai.js'
|
package/src/model.ts
ADDED
|
@@ -0,0 +1,379 @@
|
|
|
1
|
+
import { Ajv } from 'ajv'
|
|
2
|
+
import type { JSONSchemaType } from 'ajv'
|
|
3
|
+
import { AIMessage } from '@langchain/core/messages'
|
|
4
|
+
import type { AIMessageChunk, MessageFieldWithRole } from '@langchain/core/messages'
|
|
5
|
+
import type { BaseChatModel } from '@langchain/core/language_models/chat_models'
|
|
6
|
+
import { StructuredMode } from '@owlmeans/llm-common'
|
|
7
|
+
import type { NullKind } from '@owlmeans/llm-common'
|
|
8
|
+
import {
|
|
9
|
+
DEFAULT_MAX_OUTPUT_CAP, DEFAULT_MODEL_RETRIES, FALLBACK_AFTER_ATTEMPTS,
|
|
10
|
+
} from './consts.js'
|
|
11
|
+
import { LlmModelError } from './errors.js'
|
|
12
|
+
import { pluginFor, pluginOf } from './plugins/index.js'
|
|
13
|
+
import type { LlmPlugin } from './plugins/types.js'
|
|
14
|
+
import { coerceToSchema, parseJsonContent } from './helpers/json.js'
|
|
15
|
+
import { normalizeInput } from './helpers/messages.js'
|
|
16
|
+
import { withRetry } from './helpers/retry.js'
|
|
17
|
+
import { spectate } from './helpers/spectate.js'
|
|
18
|
+
import { idleTimeout, readConfig } from './utils/config.js'
|
|
19
|
+
import { reportNull } from './utils/null-report.js'
|
|
20
|
+
import type { NullReportParams } from './utils/null-report.js'
|
|
21
|
+
import { applyNoThink, ensureJsonMention } from './utils/prompt.js'
|
|
22
|
+
import { resolveSchemaValidator, toToolName, unwrapNamed } from './utils/schema.js'
|
|
23
|
+
import { streamWithDeadline } from './utils/stream.js'
|
|
24
|
+
import type {
|
|
25
|
+
LlmAskOptions, LlmInvokeOptions, LlmModel, LlmModelOptions, LlmRequestOptions,
|
|
26
|
+
LlmSpectator, LlmTalkOptions, ModelInput, RefferedResult,
|
|
27
|
+
} from './types.js'
|
|
28
|
+
|
|
29
|
+
type StreamOptions = Parameters<BaseChatModel['stream']>[1]
|
|
30
|
+
|
|
31
|
+
/**
|
|
32
|
+
* Build the four-method model API on top of a LangChain chat model.
|
|
33
|
+
*
|
|
34
|
+
* Everything provider-specific — how the client is refined between retries, how
|
|
35
|
+
* structured output is requested, whether prompt caching exists — is delegated to the
|
|
36
|
+
* `LlmPlugin` resolved for this model (see `plugins/`). The model itself only owns the
|
|
37
|
+
* provider-independent parts: streaming under an idle deadline, retry/fallback
|
|
38
|
+
* escalation, schema validation and coercion, spectator logging, and null diagnostics.
|
|
39
|
+
*/
|
|
40
|
+
export const makeLlmModel = ({
|
|
41
|
+
model,
|
|
42
|
+
outputErrors = false,
|
|
43
|
+
captureNull = false,
|
|
44
|
+
retries = DEFAULT_MODEL_RETRIES,
|
|
45
|
+
purpose,
|
|
46
|
+
}: LlmModelOptions, spectator: LlmSpectator): LlmModel => {
|
|
47
|
+
|
|
48
|
+
const ajv = new Ajv({ strict: false })
|
|
49
|
+
|
|
50
|
+
// The original config and its plugin are static per model instance: a REFINED instance
|
|
51
|
+
// is rebuilt from `lc_kwargs` and does not reliably carry the metadata back.
|
|
52
|
+
const config = readConfig(model)
|
|
53
|
+
const plugin: LlmPlugin | undefined = pluginOf(config.provider) ?? pluginFor(model)
|
|
54
|
+
const timeout = idleTimeout(config)
|
|
55
|
+
|
|
56
|
+
/** Normalize, then apply every in-place prompt adaptation, in dependency order. */
|
|
57
|
+
const prepare = (input: ModelInput, useCache: boolean, cacheMax: number, json: boolean): MessageFieldWithRole[] => {
|
|
58
|
+
const msgs = normalizeInput(input)
|
|
59
|
+
if (json) ensureJsonMention(msgs)
|
|
60
|
+
applyNoThink(msgs, config.disableThinking)
|
|
61
|
+
// Cache markers replace string content with content blocks, so they must go last.
|
|
62
|
+
if (plugin?.patchCache?.(msgs, { model, useCache, cacheMax }) === true) {
|
|
63
|
+
console.log(`Prompt caching enabled for ${plugin.type} (up to ${cacheMax} breakpoints)`)
|
|
64
|
+
}
|
|
65
|
+
return msgs
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
const notifyRef = <T>(ref: RefferedResult<T> | undefined, value: T): void => {
|
|
69
|
+
if (ref != null) {
|
|
70
|
+
ref.value = value
|
|
71
|
+
void ref.callback?.(value).finally()
|
|
72
|
+
}
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
/**
|
|
76
|
+
* Record the diagnostics of a call that produced nothing usable and build the
|
|
77
|
+
* retryable error describing it. The caller throws it, so control flow stays visible.
|
|
78
|
+
*/
|
|
79
|
+
const nullResult = async (
|
|
80
|
+
kind: NullKind,
|
|
81
|
+
p: Omit<NullReportParams, 'kind' | 'purpose' | 'config'>,
|
|
82
|
+
parsed: boolean = false,
|
|
83
|
+
): Promise<LlmModelError> => {
|
|
84
|
+
await reportNull(spectator, captureNull, { ...p, kind, purpose, config })
|
|
85
|
+
const toolCalls = (p.raw as unknown as { tool_calls?: unknown[] } | null)?.tool_calls?.length ?? 0
|
|
86
|
+
const content = JSON.stringify(p.raw?.content ?? null).substring(0, 120)
|
|
87
|
+
return new LlmModelError(
|
|
88
|
+
`null-output:raw=${p.raw != null}, parsed=${parsed}, toolCalls=${toolCalls}, content=${content}`
|
|
89
|
+
)
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
/**
|
|
93
|
+
* Rebuild the model for attempt N.
|
|
94
|
+
*
|
|
95
|
+
* Two-layer fallback: once a cheap primary has failed {@link FALLBACK_AFTER_ATTEMPTS}
|
|
96
|
+
* times, escalate to the stronger model the service attached as `__fallbackModel`. The
|
|
97
|
+
* escalation only happens WITHIN one plugin family — rotating providers mid-call would
|
|
98
|
+
* flip the structured-output call shape (tool_choice spelling, native support), so it is
|
|
99
|
+
* better to keep retrying on the primary than to switch families.
|
|
100
|
+
*/
|
|
101
|
+
const refineModel = (attempt: number, temperature?: number): BaseChatModel => {
|
|
102
|
+
const fallbackModel = (model as unknown as { __fallbackModel?: BaseChatModel }).__fallbackModel
|
|
103
|
+
const sameFamily = fallbackModel != null && plugin != null
|
|
104
|
+
&& plugin.owns(model) && plugin.owns(fallbackModel)
|
|
105
|
+
const base = (sameFamily && attempt >= FALLBACK_AFTER_ATTEMPTS) ? fallbackModel : model
|
|
106
|
+
|
|
107
|
+
if (attempt === FALLBACK_AFTER_ATTEMPTS && fallbackModel != null) {
|
|
108
|
+
if (base !== model) {
|
|
109
|
+
console.warn(`${base.getName()}: switching to fallback model after ${attempt} failed attempts`)
|
|
110
|
+
} else {
|
|
111
|
+
console.warn(
|
|
112
|
+
`Skipping cross-family fallback (${fallbackModel.getName()}); staying on ${model.getName()}`
|
|
113
|
+
)
|
|
114
|
+
}
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
const baseConfig = readConfig(base)
|
|
118
|
+
const basePlugin = pluginOf(baseConfig.provider) ?? pluginFor(base)
|
|
119
|
+
if (basePlugin == null) return base
|
|
120
|
+
|
|
121
|
+
const maxOutputCap = typeof baseConfig.maxTokensCap === 'number' && baseConfig.maxTokensCap > 0
|
|
122
|
+
? baseConfig.maxTokensCap
|
|
123
|
+
: DEFAULT_MAX_OUTPUT_CAP
|
|
124
|
+
const refined = basePlugin.refine({ base, attempt, temperature, maxOutputCap })
|
|
125
|
+
|
|
126
|
+
if (attempt > 0) {
|
|
127
|
+
const maxTokens = (refined as unknown as { maxTokens?: number }).maxTokens
|
|
128
|
+
console.warn(`${refined.getName()}: retry attempt ${attempt}, maxTokens now ${maxTokens}`)
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
return refined
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
/** How this model should be asked for schema-conforming output. */
|
|
135
|
+
const structuredMode = (): StructuredMode => plugin != null
|
|
136
|
+
? plugin.structuredMode(config as Parameters<LlmPlugin['structuredMode']>[0])
|
|
137
|
+
: (config.structuredOutput === true ? StructuredMode.Native : StructuredMode.Tool)
|
|
138
|
+
|
|
139
|
+
/**
|
|
140
|
+
* Shared structured-output core for `invoke`/`request`. Streams under the idle deadline
|
|
141
|
+
* (whose break-on-`finish_reason` also dedups the duplicate final chunk some providers
|
|
142
|
+
* emit, which would otherwise corrupt accumulated tool-call arguments), accumulates the
|
|
143
|
+
* chunks and extracts the parsed object.
|
|
144
|
+
*
|
|
145
|
+
* `withStructuredOutput` is deliberately NOT used: it buffers the whole raw stream
|
|
146
|
+
* before yielding, which is too late to dedup.
|
|
147
|
+
*/
|
|
148
|
+
const streamStructured = async <T>(
|
|
149
|
+
refined: BaseChatModel,
|
|
150
|
+
msgs: MessageFieldWithRole[],
|
|
151
|
+
innerSchema: JSONSchemaType<T>,
|
|
152
|
+
toolName: string,
|
|
153
|
+
action: string,
|
|
154
|
+
): Promise<{ piece: AIMessageChunk | null; result: T | null; mode: StructuredMode }> => {
|
|
155
|
+
const responseFormat = structuredMode() === StructuredMode.Native
|
|
156
|
+
? plugin?.responseFormat?.(toolName, innerSchema)
|
|
157
|
+
: undefined
|
|
158
|
+
// A plugin that declares Native but provides no response_format falls back to tools.
|
|
159
|
+
const mode = responseFormat != null ? StructuredMode.Native : StructuredMode.Tool
|
|
160
|
+
|
|
161
|
+
const start = (signal: AbortSignal): Promise<AsyncIterable<unknown>> => {
|
|
162
|
+
const base = { runName: action, metadata: { purpose }, signal }
|
|
163
|
+
if (mode === StructuredMode.Native) {
|
|
164
|
+
return refined.stream(msgs, { ...base, ...responseFormat && { response_format: responseFormat } } as unknown as StreamOptions)
|
|
165
|
+
}
|
|
166
|
+
// langchain converts the OpenAI-shaped tool DEFINITION for either provider, but the
|
|
167
|
+
// `tool_choice` shape is NOT converted — the plugin supplies the right spelling.
|
|
168
|
+
// eslint-disable-next-line @typescript-eslint/no-non-null-assertion
|
|
169
|
+
const bound = refined.bindTools!(
|
|
170
|
+
[{ type: 'function', function: { name: toolName, description: '', parameters: innerSchema } }],
|
|
171
|
+
{ tool_choice: plugin?.toolChoice(toolName) ?? { type: 'function', function: { name: toolName } } }
|
|
172
|
+
)
|
|
173
|
+
return bound.stream(msgs, base)
|
|
174
|
+
}
|
|
175
|
+
|
|
176
|
+
let piece: AIMessageChunk | null = null
|
|
177
|
+
for await (const rawChunk of streamWithDeadline(start, timeout)) {
|
|
178
|
+
const chunk = rawChunk as AIMessageChunk
|
|
179
|
+
piece = piece == null ? chunk : (piece.concat(chunk) as AIMessageChunk)
|
|
180
|
+
}
|
|
181
|
+
|
|
182
|
+
let result: T | null = null
|
|
183
|
+
if (mode === StructuredMode.Tool) {
|
|
184
|
+
const rawArgs = piece?.tool_calls?.[0]?.args ?? null
|
|
185
|
+
result = rawArgs != null
|
|
186
|
+
? (typeof rawArgs === 'string' ? parseJsonContent(rawArgs) as T | null : rawArgs as T)
|
|
187
|
+
: null
|
|
188
|
+
}
|
|
189
|
+
// Content fallback: some models ignore the pinned tool (or the JSON mode) and emit the
|
|
190
|
+
// schema-shaped JSON as plain content.
|
|
191
|
+
if (result == null && piece != null) {
|
|
192
|
+
const recovered = parseJsonContent(piece.content)
|
|
193
|
+
if (recovered != null) result = recovered as T
|
|
194
|
+
}
|
|
195
|
+
|
|
196
|
+
return { piece, result, mode }
|
|
197
|
+
}
|
|
198
|
+
|
|
199
|
+
const helper: LlmModel = {
|
|
200
|
+
ask: async (input, { ref, filter, action, useCache = false, cacheMax = 4 }: LlmAskOptions) => {
|
|
201
|
+
const msgs = prepare(input, useCache, cacheMax, false)
|
|
202
|
+
return withRetry({ retries, outputErrors }, async i => {
|
|
203
|
+
const refined = refineModel(i)
|
|
204
|
+
console.log('Use model to ask: ', refined.getName(), refined.lc_kwargs.model)
|
|
205
|
+
const startedAt = Date.now()
|
|
206
|
+
let result: AIMessageChunk | null = null
|
|
207
|
+
for await (const chunk of streamWithDeadline(
|
|
208
|
+
signal => refined.stream(msgs, { runName: action, metadata: { purpose }, signal }), timeout
|
|
209
|
+
)) {
|
|
210
|
+
result = result == null ? chunk : result.concat(chunk)
|
|
211
|
+
}
|
|
212
|
+
if (result == null) {
|
|
213
|
+
throw await nullResult('ask', { action, attempt: i, startedAt, refined, msgs, raw: result, useCache })
|
|
214
|
+
}
|
|
215
|
+
|
|
216
|
+
const message = new AIMessage(result)
|
|
217
|
+
let output: string | null = typeof result.content === 'string'
|
|
218
|
+
? result.content
|
|
219
|
+
: Array.isArray(result.content)
|
|
220
|
+
? result.content
|
|
221
|
+
.filter((c): c is { type: string; text: string } =>
|
|
222
|
+
typeof c === 'object' && c !== null && 'type' in c && c.type === 'text'
|
|
223
|
+
)
|
|
224
|
+
.map(c => c.text)
|
|
225
|
+
.join('') || null
|
|
226
|
+
: null
|
|
227
|
+
|
|
228
|
+
const entry = await spectate(spectator, 'ask')(msgs, message, action, i, startedAt)
|
|
229
|
+
if (ref != null) ref.spectatorEntry = entry
|
|
230
|
+
|
|
231
|
+
if (filter != null) {
|
|
232
|
+
output = await filter(output ?? '', message)
|
|
233
|
+
if (output == null) {
|
|
234
|
+
throw new LlmModelError(`filter-rejected:${JSON.stringify(message).substring(0, 50)}...`)
|
|
235
|
+
}
|
|
236
|
+
} else if (output == null || output.trim() === '') {
|
|
237
|
+
throw new LlmModelError(`empty-content:${JSON.stringify(message).substring(0, 50)}...`)
|
|
238
|
+
}
|
|
239
|
+
|
|
240
|
+
notifyRef(ref, message)
|
|
241
|
+
return output
|
|
242
|
+
})
|
|
243
|
+
},
|
|
244
|
+
|
|
245
|
+
talk: async (input, { ref, filter, action, useCache = false, cacheMax = 4 }: LlmTalkOptions) => {
|
|
246
|
+
const msgs = prepare(input, useCache, cacheMax, false)
|
|
247
|
+
return withRetry({ retries, outputErrors }, async i => {
|
|
248
|
+
const refined = refineModel(i)
|
|
249
|
+
console.log('Use model to talk: ', refined.getName(), refined.lc_kwargs.model)
|
|
250
|
+
const startedAt = Date.now()
|
|
251
|
+
let result: AIMessageChunk | null = null
|
|
252
|
+
for await (const chunk of streamWithDeadline(
|
|
253
|
+
signal => refined.stream(msgs, { runName: action, metadata: { purpose }, signal }), timeout
|
|
254
|
+
)) {
|
|
255
|
+
result = result == null ? chunk : result.concat(chunk)
|
|
256
|
+
}
|
|
257
|
+
if (result == null) {
|
|
258
|
+
throw await nullResult('talk', { action, attempt: i, startedAt, refined, msgs, raw: result, useCache })
|
|
259
|
+
}
|
|
260
|
+
|
|
261
|
+
let message: AIMessage | null = new AIMessage(result)
|
|
262
|
+
const entry = await spectate(spectator, 'talk')(msgs, message, action, i, startedAt)
|
|
263
|
+
if (ref != null) ref.spectatorEntry = entry
|
|
264
|
+
|
|
265
|
+
if (filter != null) {
|
|
266
|
+
message = await filter(message)
|
|
267
|
+
if (message == null) {
|
|
268
|
+
throw new LlmModelError('filter-rejected:talk')
|
|
269
|
+
}
|
|
270
|
+
}
|
|
271
|
+
|
|
272
|
+
notifyRef(ref, message)
|
|
273
|
+
return message
|
|
274
|
+
})
|
|
275
|
+
},
|
|
276
|
+
|
|
277
|
+
invoke: async <T>(
|
|
278
|
+
input: ModelInput,
|
|
279
|
+
schema: JSONSchemaType<T>,
|
|
280
|
+
{ temperature, ref, filter, action, useCache = false, cacheMax = 4 }: LlmInvokeOptions<T>
|
|
281
|
+
) => {
|
|
282
|
+
const msgs = prepare(input, useCache, cacheMax, true)
|
|
283
|
+
const { name, innerSchema, validate } = resolveSchemaValidator<T>(ajv, schema)
|
|
284
|
+
const toolName = toToolName((innerSchema as { title?: string }).title ?? name)
|
|
285
|
+
|
|
286
|
+
return withRetry({ retries, outputErrors }, async i => {
|
|
287
|
+
const refined = refineModel(i, temperature)
|
|
288
|
+
console.log('Use model invoke: ', refined.getName(), refined.lc_kwargs.model)
|
|
289
|
+
const startedAt = Date.now()
|
|
290
|
+
const { piece, result: collected } = await streamStructured(refined, msgs, innerSchema, toolName, action)
|
|
291
|
+
let result: T | null = collected
|
|
292
|
+
if (piece == null || result == null) {
|
|
293
|
+
throw await nullResult('invoke', {
|
|
294
|
+
action, attempt: i, startedAt, refined, msgs, raw: piece,
|
|
295
|
+
schema: { toolName, innerSchema }, useCache,
|
|
296
|
+
}, result != null)
|
|
297
|
+
}
|
|
298
|
+
|
|
299
|
+
const message = new AIMessage(piece)
|
|
300
|
+
const entry = await spectate(spectator, 'invoke')(msgs, message, action, i, startedAt)
|
|
301
|
+
if (ref != null) ref.spectatorEntry = entry
|
|
302
|
+
|
|
303
|
+
result = unwrapNamed(result, name)
|
|
304
|
+
result = coerceToSchema(result, innerSchema) as T
|
|
305
|
+
|
|
306
|
+
if (filter != null) {
|
|
307
|
+
const preFilter = result
|
|
308
|
+
result = await filter(result, message)
|
|
309
|
+
if (result == null) {
|
|
310
|
+
throw new LlmModelError(`filter-rejected:${JSON.stringify(preFilter).substring(0, 200)}`)
|
|
311
|
+
}
|
|
312
|
+
}
|
|
313
|
+
|
|
314
|
+
if (!validate(result)) {
|
|
315
|
+
const err = new LlmModelError(`validation-failed:${JSON.stringify(validate.errors)}`)
|
|
316
|
+
err.cause = validate.errors?.[0]
|
|
317
|
+
throw err
|
|
318
|
+
}
|
|
319
|
+
|
|
320
|
+
notifyRef(ref, message)
|
|
321
|
+
return result as T
|
|
322
|
+
})
|
|
323
|
+
},
|
|
324
|
+
|
|
325
|
+
request: async <T>(
|
|
326
|
+
input: ModelInput,
|
|
327
|
+
schema: JSONSchemaType<T>,
|
|
328
|
+
{ ref, filter, action, useCache = false, cacheMax = 4 }: LlmRequestOptions
|
|
329
|
+
) => {
|
|
330
|
+
const msgs = prepare(input, useCache, cacheMax, true)
|
|
331
|
+
const { name, innerSchema, validate } = resolveSchemaValidator<T>(ajv, schema)
|
|
332
|
+
const toolName = toToolName((innerSchema as { title?: string }).title ?? name)
|
|
333
|
+
|
|
334
|
+
return withRetry({ retries, outputErrors }, async i => {
|
|
335
|
+
const refined = refineModel(i)
|
|
336
|
+
console.log('Use model request: ', refined.getName(), refined.lc_kwargs.model)
|
|
337
|
+
const startedAt = Date.now()
|
|
338
|
+
const { piece, result: collected } = await streamStructured(refined, msgs, innerSchema, toolName, action)
|
|
339
|
+
let result: T | null = collected
|
|
340
|
+
if (piece == null || result == null) {
|
|
341
|
+
throw await nullResult('request', {
|
|
342
|
+
action, attempt: i, startedAt, refined, msgs, raw: piece,
|
|
343
|
+
schema: { toolName, innerSchema }, useCache,
|
|
344
|
+
}, result != null)
|
|
345
|
+
}
|
|
346
|
+
|
|
347
|
+
let message: AIMessage | null = new AIMessage(piece)
|
|
348
|
+
const entry = await spectate(spectator, 'request')(msgs, message, action, i, startedAt)
|
|
349
|
+
if (ref != null) ref.spectatorEntry = entry
|
|
350
|
+
|
|
351
|
+
// Structured output delivers the JSON as parsed tool arguments, not as message
|
|
352
|
+
// content. Surface it on `.content` so the AIMessage contract callers rely on holds.
|
|
353
|
+
result = unwrapNamed(result, name)
|
|
354
|
+
result = coerceToSchema(result, innerSchema) as T
|
|
355
|
+
message.content = JSON.stringify(result)
|
|
356
|
+
|
|
357
|
+
if (filter != null) {
|
|
358
|
+
message = await filter(message)
|
|
359
|
+
if (message == null) {
|
|
360
|
+
throw new LlmModelError('filter-rejected:request')
|
|
361
|
+
}
|
|
362
|
+
} else if (message.content == null || message.content.toString().trim() === '') {
|
|
363
|
+
throw new LlmModelError(`empty-content:${JSON.stringify(message).substring(0, 50)}...`)
|
|
364
|
+
}
|
|
365
|
+
|
|
366
|
+
if (!validate(JSON.parse(`${message.content}`))) {
|
|
367
|
+
const err = new LlmModelError(`validation-failed:${JSON.stringify(validate.errors)}`)
|
|
368
|
+
err.cause = validate.errors?.[0]
|
|
369
|
+
throw err
|
|
370
|
+
}
|
|
371
|
+
|
|
372
|
+
notifyRef(ref, message)
|
|
373
|
+
return message
|
|
374
|
+
})
|
|
375
|
+
},
|
|
376
|
+
}
|
|
377
|
+
|
|
378
|
+
return helper
|
|
379
|
+
}
|