@owlmeans/llm 0.1.14

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (147) hide show
  1. package/README.md +193 -0
  2. package/agent-meta/instructions/llm.instructions.md +66 -0
  3. package/agent-meta/manifest.json +23 -0
  4. package/agent-meta/skills/llm/SKILL.md +121 -0
  5. package/build/consts.d.ts +67 -0
  6. package/build/consts.d.ts.map +1 -0
  7. package/build/consts.js +76 -0
  8. package/build/consts.js.map +1 -0
  9. package/build/errors.d.ts +33 -0
  10. package/build/errors.d.ts.map +1 -0
  11. package/build/errors.js +52 -0
  12. package/build/errors.js.map +1 -0
  13. package/build/execution/index.d.ts +4 -0
  14. package/build/execution/index.d.ts.map +1 -0
  15. package/build/execution/index.js +3 -0
  16. package/build/execution/index.js.map +1 -0
  17. package/build/execution/service.d.ts +21 -0
  18. package/build/execution/service.d.ts.map +1 -0
  19. package/build/execution/service.js +129 -0
  20. package/build/execution/service.js.map +1 -0
  21. package/build/execution/types.d.ts +119 -0
  22. package/build/execution/types.d.ts.map +1 -0
  23. package/build/execution/types.js +2 -0
  24. package/build/execution/types.js.map +1 -0
  25. package/build/execution/utils.d.ts +28 -0
  26. package/build/execution/utils.d.ts.map +1 -0
  27. package/build/execution/utils.js +60 -0
  28. package/build/execution/utils.js.map +1 -0
  29. package/build/helpers/index.d.ts +5 -0
  30. package/build/helpers/index.d.ts.map +1 -0
  31. package/build/helpers/index.js +5 -0
  32. package/build/helpers/index.js.map +1 -0
  33. package/build/helpers/json.d.ts +24 -0
  34. package/build/helpers/json.d.ts.map +1 -0
  35. package/build/helpers/json.js +119 -0
  36. package/build/helpers/json.js.map +1 -0
  37. package/build/helpers/messages.d.ts +10 -0
  38. package/build/helpers/messages.d.ts.map +1 -0
  39. package/build/helpers/messages.js +9 -0
  40. package/build/helpers/messages.js.map +1 -0
  41. package/build/helpers/retry.d.ts +18 -0
  42. package/build/helpers/retry.d.ts.map +1 -0
  43. package/build/helpers/retry.js +57 -0
  44. package/build/helpers/retry.js.map +1 -0
  45. package/build/helpers/spectate.d.ts +8 -0
  46. package/build/helpers/spectate.d.ts.map +1 -0
  47. package/build/helpers/spectate.js +54 -0
  48. package/build/helpers/spectate.js.map +1 -0
  49. package/build/index.d.ts +13 -0
  50. package/build/index.d.ts.map +1 -0
  51. package/build/index.js +11 -0
  52. package/build/index.js.map +1 -0
  53. package/build/model.d.ts +12 -0
  54. package/build/model.d.ts.map +1 -0
  55. package/build/model.js +297 -0
  56. package/build/model.js.map +1 -0
  57. package/build/plugins/anthropic.d.ts +4 -0
  58. package/build/plugins/anthropic.d.ts.map +1 -0
  59. package/build/plugins/anthropic.js +86 -0
  60. package/build/plugins/anthropic.js.map +1 -0
  61. package/build/plugins/compatible.d.ts +18 -0
  62. package/build/plugins/compatible.d.ts.map +1 -0
  63. package/build/plugins/compatible.js +54 -0
  64. package/build/plugins/compatible.js.map +1 -0
  65. package/build/plugins/export.d.ts +6 -0
  66. package/build/plugins/export.d.ts.map +1 -0
  67. package/build/plugins/export.js +5 -0
  68. package/build/plugins/export.js.map +1 -0
  69. package/build/plugins/index.d.ts +18 -0
  70. package/build/plugins/index.d.ts.map +1 -0
  71. package/build/plugins/index.js +42 -0
  72. package/build/plugins/index.js.map +1 -0
  73. package/build/plugins/openai.d.ts +28 -0
  74. package/build/plugins/openai.d.ts.map +1 -0
  75. package/build/plugins/openai.js +89 -0
  76. package/build/plugins/openai.js.map +1 -0
  77. package/build/plugins/types.d.ts +80 -0
  78. package/build/plugins/types.d.ts.map +1 -0
  79. package/build/plugins/types.js +2 -0
  80. package/build/plugins/types.js.map +1 -0
  81. package/build/plugins/utils.d.ts +27 -0
  82. package/build/plugins/utils.d.ts.map +1 -0
  83. package/build/plugins/utils.js +33 -0
  84. package/build/plugins/utils.js.map +1 -0
  85. package/build/service.d.ts +24 -0
  86. package/build/service.d.ts.map +1 -0
  87. package/build/service.js +95 -0
  88. package/build/service.js.map +1 -0
  89. package/build/types.d.ts +190 -0
  90. package/build/types.d.ts.map +1 -0
  91. package/build/types.js +2 -0
  92. package/build/types.js.map +1 -0
  93. package/build/utils/config.d.ts +13 -0
  94. package/build/utils/config.d.ts.map +1 -0
  95. package/build/utils/config.js +15 -0
  96. package/build/utils/config.js.map +1 -0
  97. package/build/utils/null-report.d.ts +36 -0
  98. package/build/utils/null-report.d.ts.map +1 -0
  99. package/build/utils/null-report.js +84 -0
  100. package/build/utils/null-report.js.map +1 -0
  101. package/build/utils/prompt.d.ts +15 -0
  102. package/build/utils/prompt.d.ts.map +1 -0
  103. package/build/utils/prompt.js +45 -0
  104. package/build/utils/prompt.js.map +1 -0
  105. package/build/utils/schema.d.ts +20 -0
  106. package/build/utils/schema.d.ts.map +1 -0
  107. package/build/utils/schema.js +28 -0
  108. package/build/utils/schema.js.map +1 -0
  109. package/build/utils/stream.d.ts +22 -0
  110. package/build/utils/stream.d.ts.map +1 -0
  111. package/build/utils/stream.js +54 -0
  112. package/build/utils/stream.js.map +1 -0
  113. package/package.json +65 -0
  114. package/src/consts.ts +89 -0
  115. package/src/errors.ts +65 -0
  116. package/src/execution/index.ts +4 -0
  117. package/src/execution/service.ts +185 -0
  118. package/src/execution/types.ts +139 -0
  119. package/src/execution/utils.ts +79 -0
  120. package/src/helpers/index.ts +5 -0
  121. package/src/helpers/json.ts +117 -0
  122. package/src/helpers/messages.ts +12 -0
  123. package/src/helpers/retry.ts +59 -0
  124. package/src/helpers/spectate.ts +67 -0
  125. package/src/index.ts +13 -0
  126. package/src/model.ts +379 -0
  127. package/src/plugins/anthropic.ts +97 -0
  128. package/src/plugins/compatible.ts +62 -0
  129. package/src/plugins/export.ts +6 -0
  130. package/src/plugins/index.ts +53 -0
  131. package/src/plugins/openai.ts +108 -0
  132. package/src/plugins/types.ts +92 -0
  133. package/src/plugins/utils.ts +38 -0
  134. package/src/service.ts +125 -0
  135. package/src/types.ts +214 -0
  136. package/src/utils/config.ts +19 -0
  137. package/src/utils/null-report.ts +126 -0
  138. package/src/utils/prompt.ts +46 -0
  139. package/src/utils/schema.ts +35 -0
  140. package/src/utils/stream.ts +58 -0
  141. package/tests/context.ts +110 -0
  142. package/tests/execution.spec.ts +200 -0
  143. package/tests/helpers.spec.ts +192 -0
  144. package/tests/internals.spec.ts +141 -0
  145. package/tests/model.spec.ts +116 -0
  146. package/tests/plugins.spec.ts +227 -0
  147. package/tsconfig.json +19 -0
package/src/types.ts ADDED
@@ -0,0 +1,214 @@
1
+ import type { BaseChatModel } from '@langchain/core/language_models/chat_models'
2
+ import type { AIMessage, BaseMessage, MessageFieldWithRole } from '@langchain/core/messages'
3
+ import type { BaseCallbackHandler, CallbackHandlerMethods } from '@langchain/core/callbacks/base'
4
+ import type { JSONSchemaType } from 'ajv'
5
+ import type { InitializedService } from '@owlmeans/context'
6
+ import type {
7
+ LlmPurpose, ModelProvider, NullCapture, SpectatorArgument, SpectatorEntryLogged,
8
+ } from '@owlmeans/llm-common'
9
+
10
+ export type MaybeArray<T> = T | T[]
11
+
12
+ /**
13
+ * Out-of-band result channel for a model call: the caller passes a `ref` and receives
14
+ * the raw message, the spectator entry it was logged under, and an optional callback
15
+ * fired as soon as the value is known (before the caller's own await resolves).
16
+ */
17
+ export interface RefferedResult<T> {
18
+ spectatorEntry?: SpectatorEntryLogged
19
+ value?: T
20
+ callback?: (arg: T) => Promise<void>
21
+ }
22
+
23
+ export interface RetryOptions {
24
+ retries: number
25
+ outputErrors?: boolean
26
+ /**
27
+ * Abort the retry loop for this call. Return the error to throw, or `null` to keep
28
+ * retrying. Consulted in addition to the globally registered resolvers
29
+ * (`registerFatalError`) and the provider plugins' `isFatal`.
30
+ */
31
+ fatal?: (e: unknown) => Error | null
32
+ }
33
+
34
+ /** Decides whether an error must abort a retry loop instead of being retried. */
35
+ export interface FatalErrorResolver {
36
+ (e: unknown): Error | null
37
+ }
38
+
39
+ export interface LlmLogging {
40
+ /** Print every swallowed retry error to the console. */
41
+ outputErrors?: boolean
42
+ /** Hand a full {@link NullCapture} to the spectator when a call returns nothing usable. */
43
+ captureNull?: boolean
44
+ purpose: LlmPurpose
45
+ }
46
+
47
+ export interface LlmModelOptions extends LlmLogging {
48
+ model: BaseChatModel
49
+ retries?: number
50
+ }
51
+
52
+ export type ModelMessage = BaseMessage | MessageFieldWithRole
53
+ export type ModelInputItem = ModelMessage | string
54
+ export type ModelInput = MaybeArray<ModelInputItem>
55
+
56
+ export interface LlmCallOptions {
57
+ /** Short name of the operation — used as the LangChain run name and in spectator entries. */
58
+ action: string
59
+ /** Ask the provider to cache the prompt prefix (no-op for providers without prompt caching). */
60
+ useCache?: boolean
61
+ /** How many leading messages to mark as cacheable (capped by the provider's own limit). */
62
+ cacheMax?: number
63
+ }
64
+
65
+ export interface LlmAskOptions extends LlmCallOptions {
66
+ ref?: RefferedResult<AIMessage>
67
+ filter?: (output: string, result: AIMessage) => (string | null) | Promise<string | null>
68
+ }
69
+
70
+ export interface LlmTalkOptions extends LlmCallOptions {
71
+ ref?: RefferedResult<AIMessage>
72
+ filter?: (result: AIMessage) => Promise<AIMessage | null>
73
+ }
74
+
75
+ export interface LlmInvokeOptions<T> extends LlmCallOptions {
76
+ temperature?: number | undefined
77
+ ref?: RefferedResult<AIMessage>
78
+ filter?: (output: T, result: AIMessage) => (T | null) | Promise<T | null>
79
+ }
80
+
81
+ export interface LlmRequestOptions extends LlmCallOptions {
82
+ ref?: RefferedResult<AIMessage>
83
+ filter?: (result: AIMessage) => Promise<AIMessage | null>
84
+ }
85
+
86
+ /**
87
+ * The four ways to talk to a model. Every method streams under an idle deadline,
88
+ * retries with escalating output budget, logs to the spectator and validates the result.
89
+ *
90
+ * - `ask` → plain text.
91
+ * - `talk` → the raw `AIMessage`.
92
+ * - `invoke` → a schema-validated object.
93
+ * - `request` → an `AIMessage` whose `content` is the schema-validated JSON.
94
+ */
95
+ export interface LlmModel {
96
+ ask: (input: ModelInput, options: LlmAskOptions) => Promise<string>
97
+ talk: (input: ModelInput, options: LlmTalkOptions) => Promise<AIMessage>
98
+ invoke: <T>(input: ModelInput, schema: JSONSchemaType<T>, options: LlmInvokeOptions<T>) => Promise<T>
99
+ request: <T>(input: ModelInput, schema: JSONSchemaType<T>, options: LlmRequestOptions) => Promise<AIMessage>
100
+ }
101
+
102
+ /**
103
+ * Observability sink the model writes every call to. Deliberately minimal — a consumer's
104
+ * richer spectator (with `derive`, `update`, storage backends) satisfies it structurally.
105
+ */
106
+ export interface LlmSpectator {
107
+ log: (arg: SpectatorArgument) => Promise<SpectatorEntryLogged>
108
+ /** Optional sink for full diagnostics of a call that returned nothing usable. */
109
+ captureNull?: (capture: NullCapture) => Promise<void>
110
+ }
111
+
112
+ /** Resolves a model of the same role at a different temperature. */
113
+ export interface TemperatureFactory {
114
+ (temperature?: number | undefined): BaseChatModel
115
+ }
116
+
117
+ /**
118
+ * Full runtime configuration of one model alias. The JSON-safe subset that may travel
119
+ * inside an execution state is `ModelConfigPatch` (`@owlmeans/llm-common`) — this type
120
+ * additionally carries credentials and provider wiring and must never be serialized.
121
+ */
122
+ export interface ModelConfig {
123
+ provider?: ModelProvider | string
124
+ secret?: string
125
+ alias: string
126
+ /** Inherit every field of another alias in the same config list. */
127
+ preset?: string
128
+ model?: string
129
+ temperature?: number
130
+ /** Initial output-token budget per request (`max_tokens`). */
131
+ maxTokens?: number
132
+ /**
133
+ * Hard ceiling for output tokens used by the retry escalator. Each retry doubles
134
+ * `maxTokens` toward this cap; without it the escalator uses
135
+ * {@link DEFAULT_MAX_OUTPUT_CAP}, which exceeds many models' real per-request output
136
+ * limit and turns retries into 400 "max_tokens exceeds model limit" errors.
137
+ */
138
+ maxTokensCap?: number
139
+ topP?: number
140
+ baseUrl?: string
141
+ organization?: string
142
+ /** Provider-routing hint for aggregators that expose one (e.g. HuggingFace). */
143
+ inferenceProvider?: string
144
+ headers?: Record<string, string>
145
+ /**
146
+ * Suppress the model's chain-of-thought / "thinking" tokens. For the Qwen3 family
147
+ * this injects the `/no_think` soft switch into the prompt; without it those models
148
+ * routinely spend the entire output budget on hidden reasoning and return empty
149
+ * content with `finish_reason="length"`.
150
+ */
151
+ disableThinking?: boolean
152
+ /**
153
+ * OpenAI-compatible reasoning control, forwarded verbatim as the top-level `reasoning`
154
+ * request-body field. Use `{ max_tokens: N }` to hard-cap the thinking budget —
155
+ * universal, and it works even for always-reasoning models (the request `max_tokens`
156
+ * must stay strictly higher) — or `{ effort: 'none' }` to disable thinking on hybrid
157
+ * models that enumerate effort levels. `{ exclude: true }` keeps reasoning internal but
158
+ * omits it from the response.
159
+ */
160
+ reasoning?: {
161
+ effort?: 'none' | 'minimal' | 'low' | 'medium' | 'high' | 'xhigh'
162
+ max_tokens?: number
163
+ exclude?: boolean
164
+ enabled?: boolean
165
+ }
166
+ /**
167
+ * Force how `invoke`/`request` obtain structured output, overriding the provider
168
+ * plugin's default: `true` → the provider's NATIVE JSON-schema mode, `false` → the
169
+ * forced-`tool_choice` tool-calling hack. Plugins that support only one mode
170
+ * (Anthropic — tool calling) ignore this flag.
171
+ */
172
+ structuredOutput?: boolean
173
+ /**
174
+ * Idle/inactivity timeout (ms) for a streamed response. NOT a total cap — the timer
175
+ * resets on every chunk. Falls back to {@link MODEL_STREAM_TIMEOUT_MS}.
176
+ */
177
+ streamTimeout?: number
178
+ /**
179
+ * Stronger model the retry escalator switches to once the primary has failed
180
+ * {@link FALLBACK_AFTER_ATTEMPTS} times. Specified inline as a partial config merged
181
+ * over the base config, so it inherits `secret`, `headers`, etc. The escalation only
182
+ * happens when the fallback belongs to the SAME plugin family as the primary —
183
+ * rotating providers mid-call would flip the structured-output format.
184
+ */
185
+ fallback?: Partial<ModelConfig>
186
+ }
187
+
188
+ export interface LlmServiceOptions {
189
+ /** The full config list; resolved by `alias` on every `getModel` call. */
190
+ models: () => ModelConfig[]
191
+ }
192
+
193
+ /**
194
+ * Model factory and registry. Builds a `BaseChatModel` from a `ModelConfig` through the
195
+ * registered provider plugins, memoizing per alias+override so repeated resolution of a
196
+ * role does not rebuild the client.
197
+ */
198
+ export interface LlmService extends InitializedService {
199
+ models: Map<string, BaseChatModel>
200
+
201
+ callbacks: (BaseCallbackHandler | CallbackHandlerMethods)[]
202
+
203
+ addCallbacks: (callbacks: (BaseCallbackHandler | CallbackHandlerMethods)[]) => void
204
+
205
+ /** Resolve (and cache) the model registered under `alias`, with an optional patch. */
206
+ getModel: (alias: string, override?: Partial<ModelConfig>, createNew?: boolean) => BaseChatModel
207
+
208
+ /** The configured model list, as supplied by {@link LlmServiceOptions.models}. */
209
+ configs: () => ModelConfig[]
210
+ }
211
+
212
+ export interface WithLlmService {
213
+ llm: () => LlmService
214
+ }
@@ -0,0 +1,19 @@
1
+ import type { BaseChatModel } from '@langchain/core/language_models/chat_models'
2
+ import { MODEL_STREAM_TIMEOUT_MS } from '../consts.js'
3
+ import type { ModelConfig } from '../types.js'
4
+
5
+ /**
6
+ * Read the original `ModelConfig` back off a model instance. Every plugin's `build`
7
+ * passes `metadata: { config }` on the client constructor, so the config is reachable
8
+ * for any provider. Returns an empty object when unavailable — notably for a REFINED
9
+ * instance, which is rebuilt from `lc_kwargs` and does not reliably carry metadata.
10
+ * That is why callers read the config from the ORIGINAL model.
11
+ */
12
+ export const readConfig = (model: BaseChatModel): Partial<ModelConfig> => {
13
+ const meta = (model as unknown as { metadata?: { config?: Partial<ModelConfig> } }).metadata
14
+ return meta?.config ?? {}
15
+ }
16
+
17
+ /** Per-model idle stream timeout (ms), falling back to the package default. */
18
+ export const idleTimeout = (config: Partial<ModelConfig>): number =>
19
+ config.streamTimeout ?? MODEL_STREAM_TIMEOUT_MS
@@ -0,0 +1,126 @@
1
+ import util from 'util'
2
+ import type { AIMessageChunk, MessageFieldWithRole } from '@langchain/core/messages'
3
+ import type { BaseChatModel } from '@langchain/core/language_models/chat_models'
4
+ import { createIdOfLength } from '@owlmeans/basic-ids'
5
+ import type { LlmPurpose, NullCapture, NullKind } from '@owlmeans/llm-common'
6
+ import type { LlmSpectator, ModelConfig } from '../types.js'
7
+
8
+ export interface NullReportParams {
9
+ kind: NullKind
10
+ action: string
11
+ purpose?: LlmPurpose
12
+ attempt: number
13
+ startedAt: number
14
+ /** The instance that actually ran — its `lc_kwargs` carry the effective request shape. */
15
+ refined: BaseChatModel
16
+ /** The ORIGINAL model config (refined instances do not reliably keep metadata). */
17
+ config: Partial<ModelConfig>
18
+ msgs: MessageFieldWithRole[]
19
+ raw: AIMessageChunk | null
20
+ schema?: { toolName: string; innerSchema: unknown }
21
+ useCache: boolean
22
+ }
23
+
24
+ /** How many characters of each prompt message the console preview keeps. */
25
+ const PREVIEW_CHARS = 300
26
+
27
+ /**
28
+ * Assemble a complete, replayable record of a model call that returned nothing usable:
29
+ * the effective request, the raw response with its metadata, and the diagnostics that
30
+ * distinguish the common causes (budget spent on hidden reasoning vs. a refused tool
31
+ * call vs. an empty content array).
32
+ */
33
+ export const buildNullReport = (p: NullReportParams): NullCapture => {
34
+ // Read the request shape from lc_kwargs — the refined instance is rebuilt from those
35
+ // and does not always preserve `metadata`.
36
+ type Kwargs = {
37
+ model?: string
38
+ configuration?: { baseURL?: string }
39
+ modelKwargs?: { reasoning?: unknown }
40
+ topP?: number
41
+ }
42
+ const kwargs = p.refined.lc_kwargs as Kwargs
43
+ const raw = p.raw
44
+ const responseMeta = raw?.response_metadata as {
45
+ finish_reason?: string
46
+ usage?: { prompt_tokens?: number; completion_tokens?: number; reasoning_tokens?: number }
47
+ } | undefined
48
+ const usageMeta = raw?.usage_metadata as { input_tokens?: number; output_tokens?: number } | undefined
49
+ const toolCalls = (raw as unknown as { tool_calls?: unknown[] } | null)?.tool_calls
50
+
51
+ return {
52
+ meta: {
53
+ kind: p.kind,
54
+ action: p.action,
55
+ purpose: p.purpose,
56
+ attempt: p.attempt,
57
+ id: createIdOfLength(12),
58
+ timestamp: Date.now(),
59
+ elapsedMs: Date.now() - p.startedAt,
60
+ },
61
+ model: {
62
+ // The provider-side model slug (needed for replay); falls back to the config alias.
63
+ id: kwargs.model ?? p.config.model,
64
+ provider: p.config.provider,
65
+ baseUrl: kwargs.configuration?.baseURL,
66
+ maxTokens: (p.refined as unknown as { maxTokens?: number }).maxTokens,
67
+ reasoning: kwargs.modelKwargs?.reasoning,
68
+ temperature: (p.refined as unknown as { temperature?: number }).temperature,
69
+ topP: kwargs.topP,
70
+ },
71
+ request: {
72
+ messages: p.msgs as unknown[],
73
+ schema: p.schema,
74
+ useCache: p.useCache,
75
+ },
76
+ response: raw != null ? {
77
+ content: raw.content,
78
+ additional_kwargs: raw.additional_kwargs,
79
+ response_metadata: raw.response_metadata,
80
+ usage_metadata: raw.usage_metadata,
81
+ tool_calls: toolCalls,
82
+ } : null,
83
+ diagnostics: {
84
+ finishReason: responseMeta?.finish_reason,
85
+ inputTokens: usageMeta?.input_tokens ?? responseMeta?.usage?.prompt_tokens,
86
+ outputTokens: usageMeta?.output_tokens ?? responseMeta?.usage?.completion_tokens,
87
+ reasoningTokens: responseMeta?.usage?.reasoning_tokens,
88
+ contentEmpty: raw == null || raw.content === '' || raw.content == null
89
+ || (Array.isArray(raw.content) && raw.content.length === 0),
90
+ hadToolCall: Array.isArray(toolCalls) && toolCalls.length > 0,
91
+ },
92
+ }
93
+ }
94
+
95
+ /**
96
+ * Print the diagnostics of a null result, and hand the full capture to the spectator
97
+ * sink when the caller opted into capturing. A failing sink must never mask the model
98
+ * error the caller is about to throw.
99
+ */
100
+ export const reportNull = async (
101
+ spectator: LlmSpectator,
102
+ captureNull: boolean,
103
+ p: NullReportParams,
104
+ ): Promise<void> => {
105
+ const capture = buildNullReport(p)
106
+ const requestPreview = p.msgs.map(msg => {
107
+ const text = typeof msg.content === 'string' ? msg.content.substring(0, PREVIEW_CHARS) : '[complex content]'
108
+ return `[${(msg as { role?: string }).role ?? 'unknown'}] ${text}`
109
+ }).join('\n---\n')
110
+
111
+ console.error('[MODEL-NULL]', util.inspect(
112
+ {
113
+ meta: capture.meta, model: capture.model, diagnostics: capture.diagnostics,
114
+ response: capture.response, requestPreview,
115
+ },
116
+ { depth: null, maxStringLength: 2000, breakLength: 120 }
117
+ ))
118
+
119
+ if (captureNull) {
120
+ try {
121
+ await spectator.captureNull?.(capture)
122
+ } catch (e) {
123
+ console.warn('[MODEL-NULL] capture write failed', e)
124
+ }
125
+ }
126
+ }
@@ -0,0 +1,46 @@
1
+ import type { MessageFieldWithRole } from '@langchain/core/messages'
2
+ import { JSON_INSTRUCTION, NO_THINK_DIRECTIVE } from '../consts.js'
3
+
4
+ const messageMentions = (msg: MessageFieldWithRole, needle: string): boolean => {
5
+ if (typeof msg.content === 'string') return msg.content.toLowerCase().includes(needle)
6
+ if (Array.isArray(msg.content)) {
7
+ return msg.content.some(
8
+ (part: unknown) => typeof part === 'object' && part !== null && 'text' in (part as Record<string, unknown>)
9
+ && typeof (part as Record<string, string>).text === 'string'
10
+ && (part as Record<string, string>).text.toLowerCase().includes(needle)
11
+ )
12
+ }
13
+ return false
14
+ }
15
+
16
+ /** Append `text` to the last message, or push it as a new user message when that is not possible. */
17
+ const appendDirective = (msgs: MessageFieldWithRole[], text: string): void => {
18
+ const last = msgs[msgs.length - 1]
19
+ if (last != null && typeof last.content === 'string') {
20
+ last.content = `${last.content}\n${text}`
21
+ } else {
22
+ msgs.push({ role: 'user', content: text })
23
+ }
24
+ }
25
+
26
+ /**
27
+ * Ensure the word "json" appears somewhere in the prompt. Several providers refuse or
28
+ * silently ignore a JSON mode unless it does; when it is missing the JSON instruction is
29
+ * appended in place.
30
+ */
31
+ export const ensureJsonMention = (msgs: MessageFieldWithRole[]): void => {
32
+ if (msgs.some(msg => messageMentions(msg, 'json'))) return
33
+ appendDirective(msgs, JSON_INSTRUCTION)
34
+ }
35
+
36
+ /**
37
+ * Append the `/no_think` soft switch when the config asks for it. Without it,
38
+ * thinking-mode models frequently spend their entire output budget on hidden reasoning
39
+ * and return empty content with `finish_reason="length"`, which then fails every filter
40
+ * and exhausts the retries.
41
+ */
42
+ export const applyNoThink = (msgs: MessageFieldWithRole[], disableThinking: boolean | undefined): void => {
43
+ if (disableThinking !== true) return
44
+ if (msgs.some(msg => typeof msg.content === 'string' && msg.content.includes(NO_THINK_DIRECTIVE))) return
45
+ appendDirective(msgs, NO_THINK_DIRECTIVE)
46
+ }
@@ -0,0 +1,35 @@
1
+ import type { Ajv, JSONSchemaType, ValidateFunction } from 'ajv'
2
+ import { DEFAULT_TOOL_NAME } from '../consts.js'
3
+
4
+ /**
5
+ * Split the caller's schema into the optional wrapper `name` and the schema proper, and
6
+ * compile a validator for the latter. A `name` means the model is expected to answer
7
+ * `{ [name]: <object> }`; see {@link unwrapNamed}.
8
+ */
9
+ export const resolveSchemaValidator = <T>(ajv: Ajv, schema: JSONSchemaType<T>): {
10
+ name: string | undefined
11
+ innerSchema: JSONSchemaType<T>
12
+ validate: ValidateFunction<T>
13
+ } => {
14
+ const { name, ...innerSchema } = schema as JSONSchemaType<T> & { name?: string }
15
+ const validate = ajv.compile<T>(innerSchema as JSONSchemaType<T>)
16
+ return { name, innerSchema: innerSchema as JSONSchemaType<T>, validate }
17
+ }
18
+
19
+ /**
20
+ * Derive a function/tool name for tool-calling structured output. Provider tool names
21
+ * must match `^[A-Za-z0-9_-]+$`, so the schema title/name is sanitised; falls back to
22
+ * {@link DEFAULT_TOOL_NAME} when nothing usable is present.
23
+ */
24
+ export const toToolName = (raw: string | undefined): string => {
25
+ const cleaned = (raw ?? '').replace(/[^A-Za-z0-9_-]+/g, '_').replace(/^_+|_+$/g, '')
26
+ return cleaned.length > 0 ? cleaned : DEFAULT_TOOL_NAME
27
+ }
28
+
29
+ /** Unwrap `{ [name]: value }` when the schema declared a wrapper name. */
30
+ export const unwrapNamed = <T>(result: T, name: string | undefined): T => {
31
+ if (name != null && result != null && typeof result === 'object' && name in (result as Record<string, unknown>)) {
32
+ return (result as Record<string, unknown>)[name] as T
33
+ }
34
+ return result
35
+ }
@@ -0,0 +1,58 @@
1
+ import { LlmModelError } from '../errors.js'
2
+ import { MODEL_STREAM_TIMEOUT_MS } from '../consts.js'
3
+
4
+ /**
5
+ * Extract `finish_reason` from a stream chunk regardless of whether it is a plain
6
+ * `AIMessageChunk` (ask/talk) or a `{ raw, parsed }` combined chunk (structured output).
7
+ */
8
+ export const getChunkFinishReason = (chunk: unknown): string | undefined => {
9
+ const c = chunk as {
10
+ response_metadata?: { finish_reason?: string }
11
+ raw?: { response_metadata?: { finish_reason?: string } }
12
+ }
13
+ return c.response_metadata?.finish_reason ?? c.raw?.response_metadata?.finish_reason
14
+ }
15
+
16
+ /**
17
+ * Iterate a model stream under an IDLE (inactivity) deadline. `start` receives an
18
+ * `AbortSignal` to forward to `model.stream(..., { signal })`; the timer is re-armed on
19
+ * every received chunk, so it only fires after `timeoutMs` of SILENCE — a provider that
20
+ * accepted the request but stalls and never streams another token. On fire the call is
21
+ * aborted and surfaced as a retryable {@link LlmModelError} so the retry escalator moves
22
+ * on instead of hanging forever. Because the timer resets per token, long but actively
23
+ * streaming generations are never aborted.
24
+ *
25
+ * The loop also breaks after the first chunk carrying a non-empty `finish_reason`. Some
26
+ * providers send the final SSE data event twice, which makes `AIMessageChunk.concat()`
27
+ * double-append every string field (`finish_reason` becomes `'stopstop'`, the model name
28
+ * doubles) and corrupts accumulated tool-call argument strings, breaking structured-output
29
+ * parsing. Nothing meaningful arrives after `finish_reason`, so breaking there is safe.
30
+ */
31
+ export async function* streamWithDeadline<T>(
32
+ start: (signal: AbortSignal) => Promise<AsyncIterable<T>>,
33
+ timeoutMs: number = MODEL_STREAM_TIMEOUT_MS,
34
+ ): AsyncGenerator<T> {
35
+ const controller = new AbortController()
36
+ let timer: ReturnType<typeof setTimeout>
37
+ const arm = () => {
38
+ clearTimeout(timer)
39
+ timer = setTimeout(() => controller.abort(), timeoutMs)
40
+ }
41
+ arm()
42
+ try {
43
+ const stream = await start(controller.signal)
44
+ for await (const chunk of stream) {
45
+ arm() // reset the idle timer on each received token
46
+ yield chunk
47
+ const reason = getChunkFinishReason(chunk)
48
+ if (reason != null && reason !== '') break
49
+ }
50
+ } catch (e) {
51
+ if (controller.signal.aborted) {
52
+ throw new LlmModelError(`stream-stalled:no token for ${timeoutMs}ms (idle deadline)`)
53
+ }
54
+ throw e
55
+ } finally {
56
+ clearTimeout(timer!)
57
+ }
58
+ }
@@ -0,0 +1,110 @@
1
+ import { makeGates } from '@owlmeans/test'
2
+ import { ModelProvider } from '@owlmeans/llm-common'
3
+ import type { SpectatorArgument, SpectatorEntryLogged } from '@owlmeans/llm-common'
4
+ import type { LlmSpectator, ModelConfig } from '@owlmeans/llm'
5
+
6
+ /**
7
+ * Live-provider gates. An empty variable means the corresponding integration spec
8
+ * self-skips with a printed reason — never a failure. The offline specs never gate.
9
+ */
10
+ export const gates = makeGates({
11
+ openrouter: ['OPENROUTER_SECRET'],
12
+ anthropic: ['ANTHROPIC_SECRET'],
13
+ })
14
+
15
+ /** Role aliases the fixtures register — mirrors how a consumer names its own roles. */
16
+ export const Role = {
17
+ Analyst: 'analyst',
18
+ Picker: 'picker',
19
+ } as const
20
+
21
+ /**
22
+ * The OpenRouter preset the live specs resolve models through — the same shape a
23
+ * consumer builds (`Compatible` provider + baseUrl + a per-role reasoning cap), so the
24
+ * integration run exercises the real configuration path and not a special test one.
25
+ */
26
+ export const openRouterConfigs = (): ModelConfig[] => [
27
+ {
28
+ alias: Role.Analyst,
29
+ provider: ModelProvider.Compatible,
30
+ model: 'z-ai/glm-5.1:nitro',
31
+ secret: process.env.OPENROUTER_SECRET!,
32
+ baseUrl: process.env.OPENROUTER_URL ?? 'https://openrouter.ai/api/v1',
33
+ maxTokens: 8192,
34
+ maxTokensCap: 32000,
35
+ reasoning: { max_tokens: 1024 },
36
+ },
37
+ {
38
+ alias: Role.Picker,
39
+ preset: Role.Analyst,
40
+ provider: ModelProvider.Compatible,
41
+ secret: process.env.OPENROUTER_SECRET!,
42
+ },
43
+ ]
44
+
45
+ export const anthropicConfigs = (): ModelConfig[] => [
46
+ {
47
+ alias: Role.Analyst,
48
+ provider: ModelProvider.Anthropic,
49
+ model: 'claude-haiku-4-5-20251001',
50
+ secret: process.env.ANTHROPIC_SECRET!,
51
+ maxTokens: 4096,
52
+ maxTokensCap: 16000,
53
+ },
54
+ ]
55
+
56
+ /**
57
+ * A config list that needs no credentials — for offline construction/plugin specs.
58
+ * The analyst is deliberately a chat-completions model: the `gpt-5*` family goes through
59
+ * the Responses API, which rejects sampling parameters, and specs here assert on them.
60
+ */
61
+ export const offlineConfigs = (): ModelConfig[] => [
62
+ { alias: Role.Analyst, provider: ModelProvider.OpenAI, model: 'gpt-4.1-mini', secret: 'sk-test' },
63
+ {
64
+ alias: Role.Picker, provider: ModelProvider.Compatible, model: 'some/model', secret: 'sk-test',
65
+ baseUrl: 'https://openrouter.ai/api/v1',
66
+ fallback: { model: 'some/stronger-model' },
67
+ },
68
+ ]
69
+
70
+ export interface RecordingSpectator extends LlmSpectator {
71
+ entries: SpectatorEntryLogged[]
72
+ nulls: unknown[]
73
+ }
74
+
75
+ /** In-memory spectator: records what the model logged, so specs can assert on it. */
76
+ export const recordingSpectator = (): RecordingSpectator => {
77
+ const entries: SpectatorEntryLogged[] = []
78
+ const nulls: unknown[] = []
79
+ return {
80
+ entries,
81
+ nulls,
82
+ log: async (arg: SpectatorArgument) => {
83
+ const entry: SpectatorEntryLogged = {
84
+ id: `entry-${entries.length}`,
85
+ kind: 'general',
86
+ model: 'test-model',
87
+ purpose: { type: 'test' },
88
+ timestamp: entries.length,
89
+ ...arg,
90
+ }
91
+ entries.push(entry)
92
+ return entry
93
+ },
94
+ captureNull: async capture => {
95
+ nulls.push(capture)
96
+ },
97
+ }
98
+ }
99
+
100
+ /** Minimal object schema used by the structured-output specs. */
101
+ export const SpecificationSchema = {
102
+ type: 'object',
103
+ title: 'Specification',
104
+ properties: {
105
+ title: { type: 'string' },
106
+ summary: { type: 'string' },
107
+ },
108
+ required: ['title', 'summary'],
109
+ additionalProperties: true,
110
+ } as const