@owlmeans/llm 0.1.14

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (147) hide show
  1. package/README.md +193 -0
  2. package/agent-meta/instructions/llm.instructions.md +66 -0
  3. package/agent-meta/manifest.json +23 -0
  4. package/agent-meta/skills/llm/SKILL.md +121 -0
  5. package/build/consts.d.ts +67 -0
  6. package/build/consts.d.ts.map +1 -0
  7. package/build/consts.js +76 -0
  8. package/build/consts.js.map +1 -0
  9. package/build/errors.d.ts +33 -0
  10. package/build/errors.d.ts.map +1 -0
  11. package/build/errors.js +52 -0
  12. package/build/errors.js.map +1 -0
  13. package/build/execution/index.d.ts +4 -0
  14. package/build/execution/index.d.ts.map +1 -0
  15. package/build/execution/index.js +3 -0
  16. package/build/execution/index.js.map +1 -0
  17. package/build/execution/service.d.ts +21 -0
  18. package/build/execution/service.d.ts.map +1 -0
  19. package/build/execution/service.js +129 -0
  20. package/build/execution/service.js.map +1 -0
  21. package/build/execution/types.d.ts +119 -0
  22. package/build/execution/types.d.ts.map +1 -0
  23. package/build/execution/types.js +2 -0
  24. package/build/execution/types.js.map +1 -0
  25. package/build/execution/utils.d.ts +28 -0
  26. package/build/execution/utils.d.ts.map +1 -0
  27. package/build/execution/utils.js +60 -0
  28. package/build/execution/utils.js.map +1 -0
  29. package/build/helpers/index.d.ts +5 -0
  30. package/build/helpers/index.d.ts.map +1 -0
  31. package/build/helpers/index.js +5 -0
  32. package/build/helpers/index.js.map +1 -0
  33. package/build/helpers/json.d.ts +24 -0
  34. package/build/helpers/json.d.ts.map +1 -0
  35. package/build/helpers/json.js +119 -0
  36. package/build/helpers/json.js.map +1 -0
  37. package/build/helpers/messages.d.ts +10 -0
  38. package/build/helpers/messages.d.ts.map +1 -0
  39. package/build/helpers/messages.js +9 -0
  40. package/build/helpers/messages.js.map +1 -0
  41. package/build/helpers/retry.d.ts +18 -0
  42. package/build/helpers/retry.d.ts.map +1 -0
  43. package/build/helpers/retry.js +57 -0
  44. package/build/helpers/retry.js.map +1 -0
  45. package/build/helpers/spectate.d.ts +8 -0
  46. package/build/helpers/spectate.d.ts.map +1 -0
  47. package/build/helpers/spectate.js +54 -0
  48. package/build/helpers/spectate.js.map +1 -0
  49. package/build/index.d.ts +13 -0
  50. package/build/index.d.ts.map +1 -0
  51. package/build/index.js +11 -0
  52. package/build/index.js.map +1 -0
  53. package/build/model.d.ts +12 -0
  54. package/build/model.d.ts.map +1 -0
  55. package/build/model.js +297 -0
  56. package/build/model.js.map +1 -0
  57. package/build/plugins/anthropic.d.ts +4 -0
  58. package/build/plugins/anthropic.d.ts.map +1 -0
  59. package/build/plugins/anthropic.js +86 -0
  60. package/build/plugins/anthropic.js.map +1 -0
  61. package/build/plugins/compatible.d.ts +18 -0
  62. package/build/plugins/compatible.d.ts.map +1 -0
  63. package/build/plugins/compatible.js +54 -0
  64. package/build/plugins/compatible.js.map +1 -0
  65. package/build/plugins/export.d.ts +6 -0
  66. package/build/plugins/export.d.ts.map +1 -0
  67. package/build/plugins/export.js +5 -0
  68. package/build/plugins/export.js.map +1 -0
  69. package/build/plugins/index.d.ts +18 -0
  70. package/build/plugins/index.d.ts.map +1 -0
  71. package/build/plugins/index.js +42 -0
  72. package/build/plugins/index.js.map +1 -0
  73. package/build/plugins/openai.d.ts +28 -0
  74. package/build/plugins/openai.d.ts.map +1 -0
  75. package/build/plugins/openai.js +89 -0
  76. package/build/plugins/openai.js.map +1 -0
  77. package/build/plugins/types.d.ts +80 -0
  78. package/build/plugins/types.d.ts.map +1 -0
  79. package/build/plugins/types.js +2 -0
  80. package/build/plugins/types.js.map +1 -0
  81. package/build/plugins/utils.d.ts +27 -0
  82. package/build/plugins/utils.d.ts.map +1 -0
  83. package/build/plugins/utils.js +33 -0
  84. package/build/plugins/utils.js.map +1 -0
  85. package/build/service.d.ts +24 -0
  86. package/build/service.d.ts.map +1 -0
  87. package/build/service.js +95 -0
  88. package/build/service.js.map +1 -0
  89. package/build/types.d.ts +190 -0
  90. package/build/types.d.ts.map +1 -0
  91. package/build/types.js +2 -0
  92. package/build/types.js.map +1 -0
  93. package/build/utils/config.d.ts +13 -0
  94. package/build/utils/config.d.ts.map +1 -0
  95. package/build/utils/config.js +15 -0
  96. package/build/utils/config.js.map +1 -0
  97. package/build/utils/null-report.d.ts +36 -0
  98. package/build/utils/null-report.d.ts.map +1 -0
  99. package/build/utils/null-report.js +84 -0
  100. package/build/utils/null-report.js.map +1 -0
  101. package/build/utils/prompt.d.ts +15 -0
  102. package/build/utils/prompt.d.ts.map +1 -0
  103. package/build/utils/prompt.js +45 -0
  104. package/build/utils/prompt.js.map +1 -0
  105. package/build/utils/schema.d.ts +20 -0
  106. package/build/utils/schema.d.ts.map +1 -0
  107. package/build/utils/schema.js +28 -0
  108. package/build/utils/schema.js.map +1 -0
  109. package/build/utils/stream.d.ts +22 -0
  110. package/build/utils/stream.d.ts.map +1 -0
  111. package/build/utils/stream.js +54 -0
  112. package/build/utils/stream.js.map +1 -0
  113. package/package.json +65 -0
  114. package/src/consts.ts +89 -0
  115. package/src/errors.ts +65 -0
  116. package/src/execution/index.ts +4 -0
  117. package/src/execution/service.ts +185 -0
  118. package/src/execution/types.ts +139 -0
  119. package/src/execution/utils.ts +79 -0
  120. package/src/helpers/index.ts +5 -0
  121. package/src/helpers/json.ts +117 -0
  122. package/src/helpers/messages.ts +12 -0
  123. package/src/helpers/retry.ts +59 -0
  124. package/src/helpers/spectate.ts +67 -0
  125. package/src/index.ts +13 -0
  126. package/src/model.ts +379 -0
  127. package/src/plugins/anthropic.ts +97 -0
  128. package/src/plugins/compatible.ts +62 -0
  129. package/src/plugins/export.ts +6 -0
  130. package/src/plugins/index.ts +53 -0
  131. package/src/plugins/openai.ts +108 -0
  132. package/src/plugins/types.ts +92 -0
  133. package/src/plugins/utils.ts +38 -0
  134. package/src/service.ts +125 -0
  135. package/src/types.ts +214 -0
  136. package/src/utils/config.ts +19 -0
  137. package/src/utils/null-report.ts +126 -0
  138. package/src/utils/prompt.ts +46 -0
  139. package/src/utils/schema.ts +35 -0
  140. package/src/utils/stream.ts +58 -0
  141. package/tests/context.ts +110 -0
  142. package/tests/execution.spec.ts +200 -0
  143. package/tests/helpers.spec.ts +192 -0
  144. package/tests/internals.spec.ts +141 -0
  145. package/tests/model.spec.ts +116 -0
  146. package/tests/plugins.spec.ts +227 -0
  147. package/tsconfig.json +19 -0
@@ -0,0 +1,97 @@
1
+ import { ChatAnthropic } from '@langchain/anthropic'
2
+ import { BadRequestError } from '@anthropic-ai/sdk'
3
+ import type { BaseChatModel } from '@langchain/core/language_models/chat_models'
4
+ import { ModelProvider, StructuredMode } from '@owlmeans/llm-common'
5
+ import type { LlmPlugin } from './types.js'
6
+ import { MAX_CACHE_BREAKPOINTS } from '../consts.js'
7
+ import { escalateMaxTokens, makeClientOptions } from './utils.js'
8
+
9
+ /** Model-name prefix that supports prompt caching through `cache_control` markers. */
10
+ const CACHEABLE_PREFIX = 'claude-'
11
+
12
+ export const ANTHROPIC_FAMILY = 'anthropic'
13
+
14
+ export const anthropicPlugin: LlmPlugin = {
15
+ type: ModelProvider.Anthropic,
16
+
17
+ family: ANTHROPIC_FAMILY,
18
+
19
+ owns: model => model instanceof ChatAnthropic,
20
+
21
+ /**
22
+ * Anthropic has no `response_format: json_schema` mode, so structured output is
23
+ * always the forced-tool-call hack. `ModelConfig.structuredOutput` is ignored.
24
+ */
25
+ structuredMode: () => StructuredMode.Tool,
26
+
27
+ /**
28
+ * Anthropic 400s on the OpenAI spelling: "tool_choice: Input tag 'function' … does not
29
+ * match any of the expected tags: 'auto','any','tool','none'".
30
+ */
31
+ toolChoice: (toolName: string): unknown => ({ type: 'tool', name: toolName }),
32
+
33
+ build: ({ config, secret, callbacks }) => {
34
+ const model = config.model ??= 'claude-haiku-4-5-20251001'
35
+ const cfg = {
36
+ model,
37
+ apiKey: secret,
38
+ // Neither knob set → pin temperature to 0 for determinism.
39
+ ...(config.temperature == null && config.topP == null ? { temperature: 0 } : {}),
40
+ maxTokens: config.maxTokens ?? 4096,
41
+ maxRetries: 5,
42
+ metadata: { config },
43
+ callbacks,
44
+ ...(config.temperature != null ? { temperature: config.temperature } : {}),
45
+ ...(config.topP != null && config.temperature == null ? { topP: config.topP } : {}),
46
+ ...makeClientOptions({ headers: config.headers }),
47
+ }
48
+ // Anthropic rejects temperature and top_p together.
49
+ if (cfg.temperature != null && cfg.topP != null) {
50
+ delete cfg.topP
51
+ }
52
+
53
+ return new ChatAnthropic(cfg)
54
+ },
55
+
56
+ refine: ({ base, attempt, temperature, maxOutputCap }): BaseChatModel => {
57
+ const model = base as ChatAnthropic
58
+ const currentTemperature = temperature ?? model.temperature ?? 0
59
+ const maxTokens = escalateMaxTokens(model.maxTokens, attempt, maxOutputCap)
60
+ const cfg: Partial<ChatAnthropic> = {
61
+ ...(model.lc_kwargs as Partial<ChatAnthropic>), temperature: currentTemperature, maxTokens,
62
+ }
63
+ if (cfg.temperature != null && cfg.temperature > 0 && cfg.topP != null) {
64
+ delete cfg.topP
65
+ } else if (cfg.temperature != null && cfg.temperature <= 0 && cfg.topP != null) {
66
+ delete cfg.temperature
67
+ }
68
+
69
+ return new ChatAnthropic(cfg)
70
+ },
71
+
72
+ /**
73
+ * Mark the leading messages with an ephemeral `cache_control` breakpoint. Anthropic
74
+ * allows at most {@link MAX_CACHE_BREAKPOINTS}; string content is lifted into a
75
+ * single text block so the marker has somewhere to live.
76
+ */
77
+ patchCache: (msgs, { model, useCache, cacheMax }) => {
78
+ if (!useCache || !(model as ChatAnthropic).modelName?.startsWith(CACHEABLE_PREFIX)) return false
79
+ const max = Math.min(cacheMax, MAX_CACHE_BREAKPOINTS)
80
+ let i = 0
81
+ for (const msg of msgs) {
82
+ msg.content = typeof msg.content === 'string' ? [{
83
+ type: 'text',
84
+ text: msg.content,
85
+ cache_control: { type: 'ephemeral' },
86
+ }] : msg.content
87
+ if (++i > max - 1) break
88
+ }
89
+ return true
90
+ },
91
+
92
+ /**
93
+ * A malformed request (bad schema, unsupported parameter, oversized `max_tokens`)
94
+ * cannot be fixed by retrying — surface it immediately instead of burning the budget.
95
+ */
96
+ isFatal: e => e instanceof BadRequestError ? e : null,
97
+ }
@@ -0,0 +1,62 @@
1
+ import { ChatOpenAI } from '@langchain/openai'
2
+ import { ModelProvider, StructuredMode } from '@owlmeans/llm-common'
3
+ import type { LlmPlugin } from './types.js'
4
+ import type { ModelConfig } from '../types.js'
5
+ import { makeConfiguration } from './utils.js'
6
+ import { openAiFamily } from './openai.js'
7
+
8
+ /** Aggregators that encode the serving provider as a `model:provider` suffix. */
9
+ const HUGGINGFACE_MARKER = 'huggingface'
10
+
11
+ /**
12
+ * Any OpenAI-compatible endpoint that is not OpenAI itself — OpenRouter, the
13
+ * HuggingFace router, Together, a self-hosted vLLM, …
14
+ *
15
+ * Differences from the proprietary `openai` plugin:
16
+ * - Structured output defaults to the forced-tool-call hack. Native `response_format`
17
+ * support is inconsistent across the long tail of servers behind these endpoints, and
18
+ * a server that silently ignores it returns prose instead of JSON.
19
+ * - `modelKwargs` forwards aggregator-specific top-level request fields verbatim:
20
+ * `reasoning` (hard-caps/disables thinking per the config) and
21
+ * `provider.require_parameters`, which tells the aggregator to exclude servers that do
22
+ * not honour every parameter in the request — without it a request can be routed to a
23
+ * server that ignores `tool_choice` and answers `finish_reason='stop'` with no tool
24
+ * call, or that sends the final SSE chunk twice.
25
+ */
26
+ export const compatiblePlugin: LlmPlugin = {
27
+ ...openAiFamily,
28
+
29
+ type: ModelProvider.Compatible,
30
+
31
+ structuredMode: (config: ModelConfig): StructuredMode =>
32
+ config.structuredOutput === true ? StructuredMode.Native : StructuredMode.Tool,
33
+
34
+ build: ({ config, secret, callbacks }) => {
35
+ const baseModel = config.model ??= 'Qwen/Qwen3-235B-A22B-Instruct-2507'
36
+ const baseURL = config.baseUrl
37
+ const isHuggingFace = baseURL != null && baseURL.includes(HUGGINGFACE_MARKER)
38
+ const model = config.inferenceProvider != null && isHuggingFace
39
+ ? `${baseModel}:${config.inferenceProvider}`
40
+ : baseModel
41
+ const headers: Record<string, string> = {
42
+ ...(isHuggingFace && config.organization != null ? { 'X-HF-Bill-To': config.organization } : {}),
43
+ ...config.headers,
44
+ }
45
+
46
+ return new ChatOpenAI({
47
+ model,
48
+ apiKey: secret,
49
+ temperature: config.temperature ?? 0,
50
+ maxTokens: config.maxTokens ?? 4096,
51
+ topP: config.topP ?? 0.8,
52
+ maxRetries: 5,
53
+ metadata: { config },
54
+ callbacks,
55
+ modelKwargs: {
56
+ ...(config.reasoning != null ? { reasoning: config.reasoning } : {}),
57
+ provider: { require_parameters: true },
58
+ },
59
+ ...makeConfiguration({ baseURL, headers }),
60
+ })
61
+ },
62
+ }
@@ -0,0 +1,6 @@
1
+
2
+ export type * from './types.js'
3
+ export * from './anthropic.js'
4
+ export * from './compatible.js'
5
+ export * from './openai.js'
6
+ export { plugins, registerLlmPlugin, pluginOf, pluginFor, resolvePlugin } from './index.js'
@@ -0,0 +1,53 @@
1
+ import type { BaseChatModel } from '@langchain/core/language_models/chat_models'
2
+ import { LlmPluginError } from '../errors.js'
3
+ import type { LlmPlugin } from './types.js'
4
+ import { anthropicPlugin } from './anthropic.js'
5
+ import { compatiblePlugin } from './compatible.js'
6
+ import { openAiPlugin } from './openai.js'
7
+
8
+ export const plugins: Record<string, LlmPlugin> = {}
9
+
10
+ /**
11
+ * Lookup order for INSTANCE-based resolution (a model whose config metadata is not
12
+ * reachable). The first plugin whose `owns` matches wins, so the conservative member of
13
+ * a client family must come first: `compatible` precedes `openai` because both build a
14
+ * `ChatOpenAI`, and assuming the tool-calling hack for an unlabelled model is safe
15
+ * everywhere, while assuming native JSON-schema support is not.
16
+ */
17
+ const order: string[] = []
18
+
19
+ /** Register (or replace) a provider plugin. Later registrations go last in the lookup order. */
20
+ export const registerLlmPlugin = (plugin: LlmPlugin): void => {
21
+ if (plugins[plugin.type] == null) {
22
+ order.push(plugin.type)
23
+ }
24
+ plugins[plugin.type] = plugin
25
+ }
26
+
27
+ registerLlmPlugin(anthropicPlugin)
28
+ registerLlmPlugin(compatiblePlugin)
29
+ registerLlmPlugin(openAiPlugin)
30
+
31
+ /** The plugin registered for `provider`, or `undefined`. */
32
+ export const pluginOf = (provider: string | undefined): LlmPlugin | undefined =>
33
+ provider != null ? plugins[provider] : undefined
34
+
35
+ /** The first registered plugin that recognises this model instance, or `undefined`. */
36
+ export const pluginFor = (model: BaseChatModel): LlmPlugin | undefined =>
37
+ order.map(type => plugins[type]).find(plugin => plugin?.owns(model) === true)
38
+
39
+ /**
40
+ * Resolve the plugin governing a call. The config's `provider` is authoritative; when it
41
+ * is unavailable (a refined instance whose metadata did not survive) the model instance
42
+ * is matched against the registration order.
43
+ */
44
+ export const resolvePlugin = (
45
+ config: { provider?: string } | undefined,
46
+ model?: BaseChatModel,
47
+ ): LlmPlugin => {
48
+ const byType = pluginOf(config?.provider)
49
+ if (byType != null) return byType
50
+ const byModel = model != null ? pluginFor(model) : undefined
51
+ if (byModel != null) return byModel
52
+ throw new LlmPluginError(`${LlmPluginError.NO_PLUGIN}:${config?.provider ?? 'unknown'}`)
53
+ }
@@ -0,0 +1,108 @@
1
+ import { ChatOpenAI } from '@langchain/openai'
2
+ import type { BaseChatModel } from '@langchain/core/language_models/chat_models'
3
+ import { ModelProvider, StructuredMode } from '@owlmeans/llm-common'
4
+ import type { LlmPlugin, LlmRefineParams } from './types.js'
5
+ import type { ModelConfig } from '../types.js'
6
+ import { escalateMaxTokens, makeConfiguration } from './utils.js'
7
+
8
+ /** Model families served through OpenAI's Responses API rather than chat completions. */
9
+ const RESPONSES_API_PREFIXES = ['gpt-5', 'codex-']
10
+
11
+ export const OPENAI_FAMILY = 'openai'
12
+
13
+ /** Every plugin that constructs a `ChatOpenAI` shares these instance-level behaviours. */
14
+ export const openAiFamily = {
15
+ family: OPENAI_FAMILY,
16
+
17
+ owns: (model: BaseChatModel): boolean => model instanceof ChatOpenAI,
18
+
19
+ /**
20
+ * langchain converts the OpenAI-shaped tool DEFINITION for either provider, but the
21
+ * `tool_choice` shape is NOT converted — this is the OpenAI spelling.
22
+ */
23
+ toolChoice: (toolName: string): unknown => ({ type: 'function', function: { name: toolName } }),
24
+
25
+ /**
26
+ * `strict: false` keeps schemas that do not satisfy OpenAI strict-mode rules
27
+ * acceptable; the model's own ajv validation still enforces conformance afterwards.
28
+ * With `provider.require_parameters` already set, sending `response_format` also makes
29
+ * an aggregator route only to providers that actually support structured outputs.
30
+ */
31
+ responseFormat: (toolName: string, schema: unknown): Record<string, unknown> => ({
32
+ type: 'json_schema',
33
+ json_schema: { name: toolName, schema, strict: false },
34
+ }),
35
+
36
+ refine: ({ base, attempt, temperature, maxOutputCap }: LlmRefineParams): BaseChatModel => {
37
+ const model = base as ChatOpenAI
38
+ const currentTemperature = temperature ?? model.temperature ?? 0
39
+ const maxTokens = escalateMaxTokens(model.maxTokens, attempt, maxOutputCap)
40
+ const baseKwargs = model.lc_kwargs as ConstructorParameters<typeof ChatOpenAI>[0] & {
41
+ modelKwargs?: { reasoning?: { max_tokens?: number } } & Record<string, unknown>
42
+ }
43
+ // The dominant cause of an empty response is a reasoning model spending the whole
44
+ // budget on hidden thinking (finish_reason=length, empty content). The retry already
45
+ // raises maxTokens; ALSO shrink the absolute reasoning cap so the extra budget becomes
46
+ // visible output instead of more reasoning. Only touches `{ max_tokens: N }` reasoning
47
+ // configs — effort/enabled/exclude shapes are left untouched.
48
+ const reasoning = baseKwargs.modelKwargs?.reasoning
49
+ const modelKwargs = attempt > 0 && typeof reasoning?.max_tokens === 'number'
50
+ ? {
51
+ ...baseKwargs.modelKwargs,
52
+ reasoning: { ...reasoning, max_tokens: Math.max(256, Math.floor(reasoning.max_tokens / Math.pow(2, attempt))) },
53
+ }
54
+ : baseKwargs.modelKwargs
55
+
56
+ return new ChatOpenAI({
57
+ ...baseKwargs,
58
+ temperature: currentTemperature,
59
+ maxTokens,
60
+ ...(modelKwargs != null ? { modelKwargs } : {}),
61
+ })
62
+ },
63
+ }
64
+
65
+ /**
66
+ * Proprietary OpenAI endpoint. Defaults to the provider's NATIVE JSON-schema mode for
67
+ * structured output — it is reliable there, unlike on the long tail of
68
+ * OpenAI-compatible endpoints (see the `compatible` plugin).
69
+ */
70
+ export const openAiPlugin: LlmPlugin = {
71
+ ...openAiFamily,
72
+
73
+ type: ModelProvider.OpenAI,
74
+
75
+ structuredMode: (config: ModelConfig): StructuredMode =>
76
+ config.structuredOutput === false ? StructuredMode.Tool : StructuredMode.Native,
77
+
78
+ build: ({ config, secret, callbacks }) => {
79
+ const model = config.model ??= 'gpt-5.4-mini'
80
+ const configuration = makeConfiguration({ baseURL: undefined, headers: config.headers })
81
+
82
+ // The Responses API models reject `temperature`/`topP`.
83
+ if (RESPONSES_API_PREFIXES.some(prefix => model.startsWith(prefix))) {
84
+ return new ChatOpenAI({
85
+ model,
86
+ apiKey: secret,
87
+ maxTokens: config.maxTokens ?? 4096,
88
+ maxRetries: 5,
89
+ useResponsesApi: true,
90
+ metadata: { config },
91
+ callbacks,
92
+ ...configuration,
93
+ })
94
+ }
95
+
96
+ return new ChatOpenAI({
97
+ model,
98
+ apiKey: secret,
99
+ temperature: config.temperature ?? 0,
100
+ maxTokens: config.maxTokens ?? 4096,
101
+ topP: config.topP ?? 0.8,
102
+ maxRetries: 5,
103
+ metadata: { config },
104
+ callbacks,
105
+ ...configuration,
106
+ })
107
+ },
108
+ }
@@ -0,0 +1,92 @@
1
+ import type { BaseChatModel } from '@langchain/core/language_models/chat_models'
2
+ import type { BaseCallbackHandler, CallbackHandlerMethods } from '@langchain/core/callbacks/base'
3
+ import type { MessageFieldWithRole } from '@langchain/core/messages'
4
+ import type { StructuredMode } from '@owlmeans/llm-common'
5
+ import type { ModelConfig } from '../types.js'
6
+
7
+ export interface LlmBuildParams {
8
+ /** Alias the config was registered under — used only for error reporting. */
9
+ alias: string
10
+ /**
11
+ * Fully resolved config with the secret already stripped. Whatever the plugin puts
12
+ * on the client as `metadata.config` is what `readConfig` later reads back, so the
13
+ * plugin should apply its model default to this object before building.
14
+ */
15
+ config: ModelConfig
16
+ secret: string
17
+ callbacks: (BaseCallbackHandler | CallbackHandlerMethods)[]
18
+ }
19
+
20
+ export interface LlmRefineParams {
21
+ /** The model to rebuild — the primary, or the fallback once escalation kicked in. */
22
+ base: BaseChatModel
23
+ /** 0-based retry attempt; the output budget doubles with it. */
24
+ attempt: number
25
+ /** Call-site temperature override (`invoke`). */
26
+ temperature?: number | undefined
27
+ /** Hard ceiling the doubled output budget is clamped to. */
28
+ maxOutputCap: number
29
+ }
30
+
31
+ export interface LlmCacheParams {
32
+ /** The ORIGINAL (unrefined) model — refined instances do not always keep the name. */
33
+ model: BaseChatModel
34
+ useCache: boolean
35
+ cacheMax: number
36
+ }
37
+
38
+ /**
39
+ * Everything that differs between inference providers, in one replaceable unit.
40
+ *
41
+ * Registration order matters for instance-based lookup (`resolvePlugin` with no
42
+ * `provider` in the config): the FIRST plugin whose `owns` matches wins, so the
43
+ * conservative member of a family must be registered before the permissive one.
44
+ * See `plugins/index.ts`.
45
+ */
46
+ export interface LlmPlugin {
47
+ /** Provider identifier — a `ModelProvider` value, or a custom string. */
48
+ type: string
49
+
50
+ /**
51
+ * Instance behaviour group. Two plugins that construct the same client class share a
52
+ * family (e.g. `openai` and `compatible` both produce `ChatOpenAI`). The retry
53
+ * escalator refuses to switch to a fallback from a different family, because the
54
+ * structured-output call shape would change mid-call.
55
+ */
56
+ family: string
57
+
58
+ /** Construct the chat client from a resolved config. */
59
+ build: (params: LlmBuildParams) => BaseChatModel
60
+
61
+ /** Does this model instance belong to this plugin's client family? */
62
+ owns: (model: BaseChatModel) => boolean
63
+
64
+ /**
65
+ * Rebuild the model for retry attempt N: raise the output budget toward
66
+ * `maxOutputCap` and apply any provider-specific escalation (e.g. shrinking a
67
+ * reasoning budget so the extra tokens become visible output).
68
+ */
69
+ refine: (params: LlmRefineParams) => BaseChatModel
70
+
71
+ /** How this provider should be asked for schema-conforming output. */
72
+ structuredMode: (config: ModelConfig) => StructuredMode
73
+
74
+ /** Provider-specific `tool_choice` shape that pins the model to `toolName`. */
75
+ toolChoice: (toolName: string) => unknown
76
+
77
+ /** Provider-specific `response_format` for {@link StructuredMode.Native}. */
78
+ responseFormat?: (toolName: string, schema: unknown) => Record<string, unknown>
79
+
80
+ /**
81
+ * Mark leading messages as cacheable in-place. Returns `true` when caching was
82
+ * actually applied (the model may not support it). Omit for providers with no
83
+ * explicit prompt-cache markers.
84
+ */
85
+ patchCache?: (msgs: MessageFieldWithRole[], params: LlmCacheParams) => boolean
86
+
87
+ /**
88
+ * Classify a thrown error as fatal for the retry loop. Return the error to throw
89
+ * immediately, or `null` to let it be retried.
90
+ */
91
+ isFatal?: (e: unknown) => Error | null
92
+ }
@@ -0,0 +1,38 @@
1
+ /**
2
+ * Shared construction bits for the built-in plugins. Internal to `plugins/` — not
3
+ * part of the package's public surface.
4
+ */
5
+
6
+ /**
7
+ * `{ configuration: { baseURL?, defaultHeaders? } }` for the OpenAI-compatible client,
8
+ * or an empty object when there is nothing to set (the client rejects an empty
9
+ * `configuration` in some versions).
10
+ */
11
+ export const makeConfiguration = (
12
+ { baseURL, headers }: { baseURL: string | undefined, headers: Record<string, string> | undefined }
13
+ ): { configuration: Record<string, unknown> } | Record<string, never> => {
14
+ const hasBaseURL = baseURL != null
15
+ const hasHeaders = headers != null && Object.keys(headers).length > 0
16
+ if (!hasBaseURL && !hasHeaders) return {}
17
+ return {
18
+ configuration: {
19
+ ...(hasBaseURL ? { baseURL } : {}),
20
+ ...(hasHeaders ? { defaultHeaders: headers } : {}),
21
+ }
22
+ }
23
+ }
24
+
25
+ /** `{ clientOptions: { defaultHeaders } }` for the Anthropic client, or an empty object. */
26
+ export const makeClientOptions = (
27
+ { headers }: { headers: Record<string, string> | undefined }
28
+ ): { clientOptions: Record<string, unknown> } | Record<string, never> => {
29
+ if (headers == null || Object.keys(headers).length === 0) return {}
30
+ return { clientOptions: { defaultHeaders: headers } }
31
+ }
32
+
33
+ /**
34
+ * Output budget for retry attempt N: double the base budget per attempt, clamped to
35
+ * the model's hard ceiling.
36
+ */
37
+ export const escalateMaxTokens = (base: number | undefined, attempt: number, cap: number): number =>
38
+ Math.min((base ?? 2048) * Math.pow(2, attempt), cap)
package/src/service.ts ADDED
@@ -0,0 +1,125 @@
1
+ import { createService } from '@owlmeans/context'
2
+ import type { BasicConfig, BasicContext } from '@owlmeans/context'
3
+ import type { BaseChatModel } from '@langchain/core/language_models/chat_models'
4
+ import { LLM_SERVICE } from './consts.js'
5
+ import { LlmMissconfiguredError } from './errors.js'
6
+ import { resolvePlugin } from './plugins/index.js'
7
+ import type { LlmService, LlmServiceOptions, ModelConfig, WithLlmService } from './types.js'
8
+
9
+ /** The part of {@link LlmService} this package implements — see {@link llmServiceApi}. */
10
+ export type LlmServiceApi = Pick<LlmService, 'models' | 'callbacks' | 'addCallbacks' | 'getModel' | 'configs'>
11
+
12
+ /**
13
+ * Build the model-factory half of an LLM service, WITHOUT registering it as a context
14
+ * service. Spread it into your own `createService` implementation to publish extra
15
+ * methods (role-named accessors, domain helpers) alongside it:
16
+ *
17
+ * ```ts
18
+ * const service = createService<MyService>(alias, {
19
+ * ...llmServiceApi(options, () => service),
20
+ * getPicker: override => service.getModel(MyRole.Picker, override),
21
+ * } as MyService)
22
+ * ```
23
+ *
24
+ * `self` is a late-bound reference to the finished service, because the methods read
25
+ * mutable state (`models`, `callbacks`) off the registered instance rather than the
26
+ * literal.
27
+ */
28
+ export const llmServiceApi = (options: LlmServiceOptions, self: () => LlmService): LlmServiceApi => {
29
+
30
+ /** Build one client from a fully-resolved config — no preset/fallback logic here. */
31
+ const buildModel = (alias: string, config: ModelConfig): BaseChatModel => {
32
+ if (config.provider == null || config.secret == null) {
33
+ throw new LlmMissconfiguredError(alias)
34
+ }
35
+ const { secret, ...rest } = config
36
+ return resolvePlugin(rest).build({
37
+ alias, config: rest as ModelConfig, secret, callbacks: self().callbacks,
38
+ })
39
+ }
40
+
41
+ /**
42
+ * Resolve `alias` → config (inheriting a `preset`, applying `override`) and build it.
43
+ * A declared `fallback` is built as well and attached to the primary as a
44
+ * non-enumerable `__fallbackModel`, which the model's retry escalator reads. The
45
+ * fallback spec is merged OVER the primary config, so it inherits secret/headers.
46
+ */
47
+ const createModel = (alias: string, override: Partial<ModelConfig> = {}): BaseChatModel => {
48
+ const models = options.models()
49
+ const baseConfig = models.find(m => m.alias === alias)
50
+ if (baseConfig == null) {
51
+ throw new LlmMissconfiguredError(alias)
52
+ }
53
+ const config: ModelConfig = { ...baseConfig, ...override }
54
+ const preset: Partial<ModelConfig> = config.preset != null
55
+ ? { ...(models.find(m => m.alias === config.preset) ?? {}) }
56
+ : {}
57
+ if (preset.alias != null) {
58
+ delete preset.alias
59
+ }
60
+ Object.assign(config, preset)
61
+
62
+ const { fallback, ...primaryConfig } = config
63
+ const primary = buildModel(alias, primaryConfig)
64
+ if (fallback != null) {
65
+ const { fallback: _nested, ...fallbackConfig } = { ...primaryConfig, ...fallback }
66
+ Object.defineProperty(primary, '__fallbackModel', {
67
+ value: buildModel(alias, fallbackConfig),
68
+ enumerable: false,
69
+ configurable: true,
70
+ })
71
+ }
72
+
73
+ return primary
74
+ }
75
+
76
+ const cacheKey = (alias: string, override: Partial<ModelConfig> = {}): string =>
77
+ `${alias}:${JSON.stringify(override)}`
78
+
79
+ return {
80
+ models: new Map(),
81
+
82
+ callbacks: [],
83
+
84
+ addCallbacks: callbacks => {
85
+ self().callbacks.push(...callbacks)
86
+ },
87
+
88
+ configs: () => options.models(),
89
+
90
+ getModel: (alias, override = {}, createNew = false) => {
91
+ const service = self()
92
+ const key = cacheKey(alias, override)
93
+ const model = createNew || !service.models.has(key)
94
+ ? createModel(alias, override)
95
+ : service.models.get(key)!
96
+ if (!createNew) {
97
+ service.models.set(key, model)
98
+ }
99
+
100
+ return model
101
+ },
102
+ }
103
+ }
104
+
105
+ export const makeLlmService = (options: LlmServiceOptions, alias: string = LLM_SERVICE): LlmService => {
106
+ const service: LlmService = createService<LlmService>(
107
+ alias, llmServiceApi(options, () => service) as LlmService
108
+ )
109
+
110
+ return service
111
+ }
112
+
113
+ export const appendLlmService = <C extends BasicConfig, T extends BasicContext<C>>(
114
+ ctx: T,
115
+ options: LlmServiceOptions,
116
+ alias: string = LLM_SERVICE
117
+ ): T & WithLlmService => {
118
+ const context = ctx as T & WithLlmService
119
+
120
+ context.registerService(makeLlmService(options, alias))
121
+
122
+ context.llm = () => ctx.service(alias)
123
+
124
+ return context
125
+ }