@llm4ts/core 2.4.2 → 2.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/Connector.d.ts +4 -0
- package/dist/Connector.d.ts.map +1 -1
- package/dist/Connector.js +8 -3
- package/dist/Connector.js.map +1 -1
- package/dist/ConnectorConfig.d.ts.map +1 -1
- package/dist/ConnectorConfig.js +2 -0
- package/dist/ConnectorConfig.js.map +1 -1
- package/dist/ContextManagement.d.ts.map +1 -1
- package/dist/ContextManagement.js +1 -0
- package/dist/ContextManagement.js.map +1 -1
- package/dist/LabelScoring.d.ts +52 -0
- package/dist/LabelScoring.d.ts.map +1 -0
- package/dist/LabelScoring.js +139 -0
- package/dist/LabelScoring.js.map +1 -0
- package/dist/LlmService.d.ts +30 -2
- package/dist/LlmService.d.ts.map +1 -1
- package/dist/LlmService.js.map +1 -1
- package/dist/Models.d.ts +34 -2
- package/dist/Models.d.ts.map +1 -1
- package/dist/Models.js +37 -1
- package/dist/Models.js.map +1 -1
- package/dist/eval/Judge.d.ts +15 -0
- package/dist/eval/Judge.d.ts.map +1 -1
- package/dist/eval/Judge.js +50 -0
- package/dist/eval/Judge.js.map +1 -1
- package/dist/judgment/FakeJudgment.d.ts +22 -0
- package/dist/judgment/FakeJudgment.d.ts.map +1 -0
- package/dist/judgment/FakeJudgment.js +42 -0
- package/dist/judgment/FakeJudgment.js.map +1 -0
- package/dist/judgment/Judgment.d.ts +57 -0
- package/dist/judgment/Judgment.d.ts.map +1 -0
- package/dist/judgment/Judgment.js +57 -0
- package/dist/judgment/Judgment.js.map +1 -0
- package/dist/judgment/LlmJudgment.d.ts +73 -0
- package/dist/judgment/LlmJudgment.d.ts.map +1 -0
- package/dist/judgment/LlmJudgment.js +239 -0
- package/dist/judgment/LlmJudgment.js.map +1 -0
- package/dist/judgment/Schemas.d.ts +163 -0
- package/dist/judgment/Schemas.d.ts.map +1 -0
- package/dist/judgment/Schemas.js +197 -0
- package/dist/judgment/Schemas.js.map +1 -0
- package/dist/judgment/TypeSafeJudgment.d.ts +46 -0
- package/dist/judgment/TypeSafeJudgment.d.ts.map +1 -0
- package/dist/judgment/TypeSafeJudgment.js +185 -0
- package/dist/judgment/TypeSafeJudgment.js.map +1 -0
- package/dist/observability/MeteredLlmService.d.ts.map +1 -1
- package/dist/observability/MeteredLlmService.js +1 -0
- package/dist/observability/MeteredLlmService.js.map +1 -1
- package/dist/providers/ConnectorFactories.d.ts.map +1 -1
- package/dist/providers/ConnectorFactories.js +2 -0
- package/dist/providers/ConnectorFactories.js.map +1 -1
- package/dist/providers/LmStudioProvider.d.ts.map +1 -1
- package/dist/providers/LmStudioProvider.js +45 -34
- package/dist/providers/LmStudioProvider.js.map +1 -1
- package/dist/providers/MlxLmProvider.d.ts +29 -0
- package/dist/providers/MlxLmProvider.d.ts.map +1 -0
- package/dist/providers/MlxLmProvider.js +249 -0
- package/dist/providers/MlxLmProvider.js.map +1 -0
- package/dist/providers/MockProvider.d.ts.map +1 -1
- package/dist/providers/MockProvider.js +11 -0
- package/dist/providers/MockProvider.js.map +1 -1
- package/dist/providers/OpenAIModels.d.ts +35 -0
- package/dist/providers/OpenAIModels.d.ts.map +1 -1
- package/dist/providers/OpenAIModels.js +36 -3
- package/dist/providers/OpenAIModels.js.map +1 -1
- package/package.json +8 -1
- package/src/Connector.ts +13 -3
- package/src/ConnectorConfig.ts +2 -0
- package/src/ContextManagement.ts +1 -0
- package/src/LabelScoring.ts +211 -0
- package/src/LlmService.ts +37 -1
- package/src/Models.ts +43 -1
- package/src/eval/Judge.ts +74 -0
- package/src/judgment/FakeJudgment.ts +90 -0
- package/src/judgment/Judgment.ts +123 -0
- package/src/judgment/LlmJudgment.ts +399 -0
- package/src/judgment/Schemas.ts +278 -0
- package/src/judgment/TypeSafeJudgment.ts +253 -0
- package/src/observability/MeteredLlmService.ts +7 -0
- package/src/providers/ConnectorFactories.ts +4 -0
- package/src/providers/LmStudioProvider.ts +64 -48
- package/src/providers/MlxLmProvider.ts +357 -0
- package/src/providers/MockProvider.ts +19 -0
- package/src/providers/OpenAIModels.ts +40 -3
|
@@ -0,0 +1,357 @@
|
|
|
1
|
+
import * as Effect from "effect/Effect"
|
|
2
|
+
import * as Schema from "effect/Schema"
|
|
3
|
+
import * as Stream from "effect/Stream"
|
|
4
|
+
import { makeApiConnector, type ApiConnectorShape } from "../Connector.ts"
|
|
5
|
+
import { ConfigError, InvalidRequestError, ParseError, type LlmError } from "../Errors.ts"
|
|
6
|
+
import type { HttpClientShape } from "../HttpClient.ts"
|
|
7
|
+
import { normalizeLabelProbabilities } from "../LabelScoring.ts"
|
|
8
|
+
import type { LabelSequence, StructuredResult } from "../LlmService.ts"
|
|
9
|
+
import * as Result from "effect/Result"
|
|
10
|
+
import {
|
|
11
|
+
ConnectorCapabilities,
|
|
12
|
+
ConnectorIds,
|
|
13
|
+
LabelDistribution,
|
|
14
|
+
LlmChunk,
|
|
15
|
+
TokenUsage,
|
|
16
|
+
type JsonSchema,
|
|
17
|
+
type LlmConfig
|
|
18
|
+
} from "../Models.ts"
|
|
19
|
+
import { parseFromText, withSchemaHint } from "../StructuredOutput.ts"
|
|
20
|
+
import {
|
|
21
|
+
OpenAIChatChunk,
|
|
22
|
+
OpenAIChatCompletionRequest,
|
|
23
|
+
OpenAIChatCompletionResponse,
|
|
24
|
+
OpenAIChatMessage,
|
|
25
|
+
type OpenAITokenLogprob,
|
|
26
|
+
type OpenAITopLogprob
|
|
27
|
+
} from "./OpenAIModels.ts"
|
|
28
|
+
import { openAIHistoryMessages } from "./OpenAIProvider.ts"
|
|
29
|
+
|
|
30
|
+
/**
|
|
31
|
+
* `mlx-lm` (`python -m mlx_lm.server`): an OpenAI-compatible local server on
|
|
32
|
+
* Apple Silicon that returns token log-probabilities. That makes it the
|
|
33
|
+
* reference backend for label scoring (ADR 0017): one forward pass, one
|
|
34
|
+
* output token, the distribution read off the labels. Streaming reuses the
|
|
35
|
+
* OpenAI wire format; structured output is prompt-coerced because the
|
|
36
|
+
* server has no grammar mode; tool calling is unsupported.
|
|
37
|
+
*
|
|
38
|
+
* Verified against mlx-lm 0.31.3 on 2026-09-19: `top_logprobs` is capped
|
|
39
|
+
* at 11 and entries carry the token text, so labels match on text.
|
|
40
|
+
*/
|
|
41
|
+
|
|
42
|
+
const emptyHeaders: Readonly<Record<string, string>> = Object.freeze({})
|
|
43
|
+
|
|
44
|
+
/** The server's hard cap on `top_logprobs`. */
|
|
45
|
+
export const mlxLmTopLogprobs = 11
|
|
46
|
+
|
|
47
|
+
export const normalizeMlxLmBaseUrl = (raw: string): string => {
|
|
48
|
+
const trimmed = raw.trim().replace(/\/+$/, "")
|
|
49
|
+
return trimmed.endsWith("/v1") ? trimmed.slice(0, -"/v1".length) : trimmed
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
/**
|
|
53
|
+
* Byte-level BPE vocabularies spell a leading space as `Ġ` (GPT/Qwen) or
|
|
54
|
+
* `▁` (SentencePiece); a label token may also carry real whitespace or a
|
|
55
|
+
* different case. All of those are the same label.
|
|
56
|
+
*/
|
|
57
|
+
export const normalizeLabelToken = (token: string): string =>
|
|
58
|
+
token
|
|
59
|
+
.replace(/^[Ġ▁]+/, "")
|
|
60
|
+
.trim()
|
|
61
|
+
.toLowerCase()
|
|
62
|
+
|
|
63
|
+
/** Sum the probability mass of every top-k entry that spells one of the labels. */
|
|
64
|
+
export const labelMassFrom = (
|
|
65
|
+
labels: ReadonlyArray<string>,
|
|
66
|
+
top: ReadonlyArray<OpenAITopLogprob>
|
|
67
|
+
): Record<string, number> => {
|
|
68
|
+
const byNormalized = new Map(labels.map((label) => [normalizeLabelToken(label), label]))
|
|
69
|
+
const mass: Record<string, number> = {}
|
|
70
|
+
for (const entry of top) {
|
|
71
|
+
const label = byNormalized.get(normalizeLabelToken(entry.token))
|
|
72
|
+
if (label !== undefined) {
|
|
73
|
+
mass[label] = (mass[label] ?? 0) + Math.exp(entry.logprob)
|
|
74
|
+
}
|
|
75
|
+
}
|
|
76
|
+
return mass
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
const decodeCompletion = (raw: string): Effect.Effect<OpenAIChatCompletionResponse, ParseError> =>
|
|
80
|
+
Schema.decodeUnknownEffect(Schema.fromJsonString(OpenAIChatCompletionResponse))(raw).pipe(
|
|
81
|
+
Effect.mapError((error) =>
|
|
82
|
+
ParseError.make({
|
|
83
|
+
message: `Failed to decode mlx-lm chat completion: ${String(error)}`,
|
|
84
|
+
raw
|
|
85
|
+
})
|
|
86
|
+
)
|
|
87
|
+
)
|
|
88
|
+
|
|
89
|
+
const decodeStreamChunk = (raw: string): Effect.Effect<OpenAIChatChunk, ParseError> =>
|
|
90
|
+
Schema.decodeUnknownEffect(Schema.fromJsonString(OpenAIChatChunk))(raw).pipe(
|
|
91
|
+
Effect.mapError((error) =>
|
|
92
|
+
ParseError.make({
|
|
93
|
+
message: `Failed to parse mlx-lm stream chunk: ${String(error)}`,
|
|
94
|
+
raw
|
|
95
|
+
})
|
|
96
|
+
)
|
|
97
|
+
)
|
|
98
|
+
|
|
99
|
+
const usageOf = (response: OpenAIChatCompletionResponse): TokenUsage | undefined =>
|
|
100
|
+
response.usage === undefined
|
|
101
|
+
? undefined
|
|
102
|
+
: TokenUsage.make({
|
|
103
|
+
prompt: response.usage.prompt_tokens ?? 0,
|
|
104
|
+
completion: response.usage.completion_tokens ?? 0,
|
|
105
|
+
total:
|
|
106
|
+
response.usage.total_tokens ??
|
|
107
|
+
(response.usage.prompt_tokens ?? 0) + (response.usage.completion_tokens ?? 0)
|
|
108
|
+
})
|
|
109
|
+
|
|
110
|
+
/**
|
|
111
|
+
* Read a shared-prefix reply, one `<n>: <label>` line per question, from
|
|
112
|
+
* the generated tokens. The text is rebuilt token by token; whenever it
|
|
113
|
+
* ends in a `<n>:` marker for the next expected question, the first
|
|
114
|
+
* following token that spells one of that question's labels is that
|
|
115
|
+
* question's answer position and its top-k is the distribution. A question
|
|
116
|
+
* whose marker or label never appears fails on its own.
|
|
117
|
+
*/
|
|
118
|
+
export const labelSequenceFrom = (
|
|
119
|
+
labelSets: ReadonlyArray<ReadonlyArray<string>>,
|
|
120
|
+
tokens: ReadonlyArray<OpenAITokenLogprob>
|
|
121
|
+
): ReadonlyArray<Result.Result<LabelDistribution, ParseError>> => {
|
|
122
|
+
const found: Array<LabelDistribution | undefined> = labelSets.map(() => undefined)
|
|
123
|
+
let buffer = ""
|
|
124
|
+
let expecting = 0
|
|
125
|
+
let awaitingLabel = false
|
|
126
|
+
for (const token of tokens) {
|
|
127
|
+
const text = token.token.replace(/[Ġ▁]/g, " ")
|
|
128
|
+
if (awaitingLabel) {
|
|
129
|
+
const labels = labelSets[expecting] ?? []
|
|
130
|
+
const normalized = normalizeLabelToken(token.token)
|
|
131
|
+
if (normalized.length === 0) {
|
|
132
|
+
continue
|
|
133
|
+
}
|
|
134
|
+
if (labels.some((label) => normalizeLabelToken(label) === normalized)) {
|
|
135
|
+
const top = token.top_logprobs ?? [{ token: token.token, logprob: token.logprob }]
|
|
136
|
+
const mass = labelMassFrom(labels, top)
|
|
137
|
+
const total = Object.values(mass).reduce((sum, value) => sum + value, 0)
|
|
138
|
+
if (total > 0) {
|
|
139
|
+
found[expecting] = LabelDistribution.make({
|
|
140
|
+
probabilities: Object.fromEntries(
|
|
141
|
+
Object.entries(mass).map(([label, value]) => [label, value / total])
|
|
142
|
+
),
|
|
143
|
+
method: "logprobs",
|
|
144
|
+
support: Math.min(1, total)
|
|
145
|
+
})
|
|
146
|
+
}
|
|
147
|
+
expecting += 1
|
|
148
|
+
awaitingLabel = false
|
|
149
|
+
buffer = ""
|
|
150
|
+
continue
|
|
151
|
+
}
|
|
152
|
+
// Anything else before the label: keep scanning, but a new marker
|
|
153
|
+
// means the answer for the previous question was skipped.
|
|
154
|
+
}
|
|
155
|
+
buffer += text
|
|
156
|
+
const marker = /(\d+)\s*[:.)]\s*$/.exec(buffer)
|
|
157
|
+
if (marker !== null) {
|
|
158
|
+
const number = Number.parseInt(marker[1] ?? "", 10) - 1
|
|
159
|
+
if (number >= expecting && number < labelSets.length) {
|
|
160
|
+
expecting = number
|
|
161
|
+
awaitingLabel = true
|
|
162
|
+
buffer = ""
|
|
163
|
+
}
|
|
164
|
+
}
|
|
165
|
+
}
|
|
166
|
+
return labelSets.map((_, index) => {
|
|
167
|
+
const distribution = found[index]
|
|
168
|
+
return distribution === undefined
|
|
169
|
+
? Result.fail(
|
|
170
|
+
ParseError.make({
|
|
171
|
+
message: `question ${index + 1} of ${labelSets.length}: no label read in the batched reply`,
|
|
172
|
+
raw: tokens
|
|
173
|
+
.map((token) => token.token)
|
|
174
|
+
.join("")
|
|
175
|
+
.slice(0, 200)
|
|
176
|
+
})
|
|
177
|
+
)
|
|
178
|
+
: Result.succeed(distribution)
|
|
179
|
+
})
|
|
180
|
+
}
|
|
181
|
+
|
|
182
|
+
export const mlxLmCapabilities = (): ConnectorCapabilities =>
|
|
183
|
+
ConnectorCapabilities.make({
|
|
184
|
+
readOnlyEnforcement: "enforced",
|
|
185
|
+
labelProbabilities: "logprobs"
|
|
186
|
+
})
|
|
187
|
+
|
|
188
|
+
export const makeMlxLmProvider = (
|
|
189
|
+
config: LlmConfig,
|
|
190
|
+
httpClient: HttpClientShape
|
|
191
|
+
): ApiConnectorShape => {
|
|
192
|
+
const baseUrl: Effect.Effect<string, ConfigError> =
|
|
193
|
+
config.baseUrl === undefined
|
|
194
|
+
? Effect.fail(ConfigError.make({ message: "Missing baseUrl for mlx-lm provider" }))
|
|
195
|
+
: Effect.succeed(normalizeMlxLmBaseUrl(config.baseUrl))
|
|
196
|
+
|
|
197
|
+
const complete = (
|
|
198
|
+
request: OpenAIChatCompletionRequest
|
|
199
|
+
): Effect.Effect<OpenAIChatCompletionResponse, LlmError> =>
|
|
200
|
+
Effect.gen(function* () {
|
|
201
|
+
const normalized = yield* baseUrl
|
|
202
|
+
const raw = yield* httpClient.postJson(
|
|
203
|
+
`${normalized}/v1/chat/completions`,
|
|
204
|
+
JSON.stringify(request),
|
|
205
|
+
emptyHeaders,
|
|
206
|
+
config.timeout
|
|
207
|
+
)
|
|
208
|
+
return yield* decodeCompletion(raw)
|
|
209
|
+
})
|
|
210
|
+
|
|
211
|
+
const streamRequest = (
|
|
212
|
+
messages: ReadonlyArray<OpenAIChatMessage>
|
|
213
|
+
): Stream.Stream<LlmChunk, LlmError> =>
|
|
214
|
+
Stream.unwrap(
|
|
215
|
+
Effect.map(baseUrl, (normalized) => {
|
|
216
|
+
const request = OpenAIChatCompletionRequest.make({
|
|
217
|
+
model: config.model,
|
|
218
|
+
messages,
|
|
219
|
+
temperature: config.temperature ?? 0.7,
|
|
220
|
+
stream: true,
|
|
221
|
+
...(config.maxTokens === undefined ? {} : { max_tokens: config.maxTokens })
|
|
222
|
+
})
|
|
223
|
+
return httpClient
|
|
224
|
+
.postJsonStreamSSE(
|
|
225
|
+
`${normalized}/v1/chat/completions`,
|
|
226
|
+
JSON.stringify(request),
|
|
227
|
+
emptyHeaders,
|
|
228
|
+
config.timeout
|
|
229
|
+
)
|
|
230
|
+
.pipe(
|
|
231
|
+
Stream.mapEffect(decodeStreamChunk),
|
|
232
|
+
Stream.flatMap((chunk) => {
|
|
233
|
+
const choice = chunk.choices[0]
|
|
234
|
+
const delta = choice?.delta?.content ?? ""
|
|
235
|
+
const finishReason = choice?.finish_reason ?? undefined
|
|
236
|
+
return delta.length > 0 || finishReason !== undefined
|
|
237
|
+
? Stream.succeed(
|
|
238
|
+
LlmChunk.make({
|
|
239
|
+
delta,
|
|
240
|
+
metadata: { provider: "mlx-lm", model: config.model },
|
|
241
|
+
...(finishReason === undefined ? {} : { finishReason })
|
|
242
|
+
})
|
|
243
|
+
)
|
|
244
|
+
: Stream.empty
|
|
245
|
+
})
|
|
246
|
+
)
|
|
247
|
+
})
|
|
248
|
+
)
|
|
249
|
+
|
|
250
|
+
const executeStructuredWithUsage = <A, E, RD, RE>(
|
|
251
|
+
prompt: string,
|
|
252
|
+
schema: Schema.ConstraintCodec<A, E, RD, RE>,
|
|
253
|
+
jsonSchema: JsonSchema
|
|
254
|
+
): Effect.Effect<StructuredResult<A>, LlmError, RD> =>
|
|
255
|
+
Effect.gen(function* () {
|
|
256
|
+
const response = yield* complete(
|
|
257
|
+
OpenAIChatCompletionRequest.make({
|
|
258
|
+
model: config.model,
|
|
259
|
+
messages: [
|
|
260
|
+
OpenAIChatMessage.make({ role: "user", content: withSchemaHint(prompt, jsonSchema) })
|
|
261
|
+
],
|
|
262
|
+
temperature: config.temperature ?? 0.7,
|
|
263
|
+
stream: false,
|
|
264
|
+
...(config.maxTokens === undefined ? {} : { max_tokens: config.maxTokens })
|
|
265
|
+
})
|
|
266
|
+
)
|
|
267
|
+
const content = response.choices[0]?.message?.content?.trim() ?? ""
|
|
268
|
+
const value = yield* parseFromText(content, schema, jsonSchema)
|
|
269
|
+
const result: StructuredResult<A> = [value, usageOf(response), response.model]
|
|
270
|
+
return result
|
|
271
|
+
})
|
|
272
|
+
|
|
273
|
+
/** One forward pass: the first generated position's top-k over the labels. */
|
|
274
|
+
const scoreLabels = (
|
|
275
|
+
prompt: string,
|
|
276
|
+
labels: ReadonlyArray<string>
|
|
277
|
+
): Effect.Effect<LabelDistribution, LlmError> =>
|
|
278
|
+
Effect.gen(function* () {
|
|
279
|
+
const response = yield* complete(
|
|
280
|
+
OpenAIChatCompletionRequest.make({
|
|
281
|
+
model: config.model,
|
|
282
|
+
messages: [OpenAIChatMessage.make({ role: "user", content: prompt })],
|
|
283
|
+
temperature: 0,
|
|
284
|
+
max_tokens: 1,
|
|
285
|
+
stream: false,
|
|
286
|
+
logprobs: true,
|
|
287
|
+
top_logprobs: mlxLmTopLogprobs
|
|
288
|
+
})
|
|
289
|
+
)
|
|
290
|
+
const top = response.choices[0]?.logprobs?.content?.[0]?.top_logprobs
|
|
291
|
+
if (top === undefined) {
|
|
292
|
+
return yield* ParseError.make({
|
|
293
|
+
message: "mlx-lm returned no logprobs; is the server started with a text model?",
|
|
294
|
+
raw: JSON.stringify(response.choices[0]?.message?.content ?? "")
|
|
295
|
+
})
|
|
296
|
+
}
|
|
297
|
+
const usage = usageOf(response)
|
|
298
|
+
return yield* normalizeLabelProbabilities(labels, labelMassFrom(labels, top), "logprobs", {
|
|
299
|
+
...(usage === undefined ? {} : { usage }),
|
|
300
|
+
...(response.model === undefined ? {} : { model: response.model })
|
|
301
|
+
})
|
|
302
|
+
})
|
|
303
|
+
|
|
304
|
+
/** One call for several questions: enough tokens for one short line each, every position read. */
|
|
305
|
+
const scoreLabelSequence = (
|
|
306
|
+
prompt: string,
|
|
307
|
+
labelSets: ReadonlyArray<ReadonlyArray<string>>
|
|
308
|
+
): Effect.Effect<LabelSequence, LlmError> =>
|
|
309
|
+
Effect.gen(function* () {
|
|
310
|
+
const response = yield* complete(
|
|
311
|
+
OpenAIChatCompletionRequest.make({
|
|
312
|
+
model: config.model,
|
|
313
|
+
messages: [OpenAIChatMessage.make({ role: "user", content: prompt })],
|
|
314
|
+
temperature: 0,
|
|
315
|
+
max_tokens: labelSets.length * 8 + 4,
|
|
316
|
+
stream: false,
|
|
317
|
+
logprobs: true,
|
|
318
|
+
top_logprobs: mlxLmTopLogprobs
|
|
319
|
+
})
|
|
320
|
+
)
|
|
321
|
+
const tokens = response.choices[0]?.logprobs?.content
|
|
322
|
+
if (tokens === undefined || tokens === null) {
|
|
323
|
+
return yield* ParseError.make({
|
|
324
|
+
message: "mlx-lm returned no logprobs; is the server started with a text model?",
|
|
325
|
+
raw: JSON.stringify(response.choices[0]?.message?.content ?? "")
|
|
326
|
+
})
|
|
327
|
+
}
|
|
328
|
+
const usage = usageOf(response)
|
|
329
|
+
const sequence: LabelSequence = {
|
|
330
|
+
entries: labelSequenceFrom(labelSets, tokens),
|
|
331
|
+
...(usage === undefined ? {} : { usage }),
|
|
332
|
+
...(response.model === undefined ? {} : { model: response.model })
|
|
333
|
+
}
|
|
334
|
+
return sequence
|
|
335
|
+
})
|
|
336
|
+
|
|
337
|
+
const isAvailable: Effect.Effect<boolean> =
|
|
338
|
+
config.baseUrl === undefined
|
|
339
|
+
? Effect.succeed(false)
|
|
340
|
+
: httpClient
|
|
341
|
+
.get(`${normalizeMlxLmBaseUrl(config.baseUrl)}/v1/models`, emptyHeaders, config.timeout)
|
|
342
|
+
.pipe(Effect.match({ onFailure: () => false, onSuccess: () => true }))
|
|
343
|
+
|
|
344
|
+
return makeApiConnector({
|
|
345
|
+
id: ConnectorIds.MlxLm,
|
|
346
|
+
executeStream: (prompt) =>
|
|
347
|
+
streamRequest([OpenAIChatMessage.make({ role: "user", content: prompt })]),
|
|
348
|
+
executeStreamWithHistory: (messages) => streamRequest(openAIHistoryMessages(messages)),
|
|
349
|
+
executeWithTools: () =>
|
|
350
|
+
Effect.fail(InvalidRequestError.make({ message: "mlx-lm does not support tool calling" })),
|
|
351
|
+
executeStructuredWithUsage,
|
|
352
|
+
scoreLabels,
|
|
353
|
+
scoreLabelSequence,
|
|
354
|
+
isAvailable,
|
|
355
|
+
capabilities: mlxLmCapabilities()
|
|
356
|
+
})
|
|
357
|
+
}
|
|
@@ -16,6 +16,7 @@ import {
|
|
|
16
16
|
type LlmConfig,
|
|
17
17
|
type Message
|
|
18
18
|
} from "../Models.ts"
|
|
19
|
+
import { normalizeLabelProbabilities } from "../LabelScoring.ts"
|
|
19
20
|
import { parseFromText } from "../StructuredOutput.ts"
|
|
20
21
|
|
|
21
22
|
export const mockResponse = (): string =>
|
|
@@ -226,6 +227,10 @@ const requestedIssueCount = (prompt: string): number => {
|
|
|
226
227
|
export const mockStructuredResponse = (prompt: string, schema: JsonSchema): string => {
|
|
227
228
|
const schemaText = JSON.stringify(schema)
|
|
228
229
|
if (schemaText.includes('"summary"') && schemaText.includes('"issues"')) {
|
|
230
|
+
// Review findings carry severity; issue drafts use the template shape below.
|
|
231
|
+
if (schemaText.includes('"severity"')) {
|
|
232
|
+
return JSON.stringify({ summary: "Mock review: no issues.", issues: [] })
|
|
233
|
+
}
|
|
229
234
|
const count = requestedIssueCount(prompt)
|
|
230
235
|
return JSON.stringify({
|
|
231
236
|
summary: `Generated ${count} issue drafts for Spring Boot microservice demo.`,
|
|
@@ -317,6 +322,20 @@ export const makeMockProvider = (config: LlmConfig): ApiConnectorShape => {
|
|
|
317
322
|
): Effect.Effect<A, LlmError, RD> =>
|
|
318
323
|
Effect.map(executeStructuredWithUsage(prompt, schema, jsonSchema), ([value]) => value),
|
|
319
324
|
executeStructuredWithUsage,
|
|
325
|
+
// Deterministic: the first offered label gets most of the mass, the rest
|
|
326
|
+
// share the remainder evenly, so tests can assert on ordering.
|
|
327
|
+
scoreLabels: (_prompt, labels) =>
|
|
328
|
+
normalizeLabelProbabilities(
|
|
329
|
+
labels,
|
|
330
|
+
Object.fromEntries(
|
|
331
|
+
labels.map((label, index) => [
|
|
332
|
+
label,
|
|
333
|
+
index === 0 ? 0.6 : 0.4 / Math.max(1, labels.length - 1)
|
|
334
|
+
])
|
|
335
|
+
),
|
|
336
|
+
"verbalized",
|
|
337
|
+
{ support: 1 }
|
|
338
|
+
),
|
|
320
339
|
isAvailable: Effect.succeed(true)
|
|
321
340
|
}
|
|
322
341
|
}
|
|
@@ -6,7 +6,19 @@ export class OpenAIJsonSchemaSpec extends Schema.Class<OpenAIJsonSchemaSpec>(
|
|
|
6
6
|
"OpenAIJsonSchemaSpec"
|
|
7
7
|
)({
|
|
8
8
|
name: Schema.String,
|
|
9
|
-
schema: JsonSchema
|
|
9
|
+
schema: JsonSchema,
|
|
10
|
+
strict: Schema.optionalKey(Schema.Boolean)
|
|
11
|
+
}) {}
|
|
12
|
+
|
|
13
|
+
/**
|
|
14
|
+
* Chat-template switches forwarded verbatim by OpenAI-compatible local
|
|
15
|
+
* servers (LM Studio, mlx-lm). `enable_thinking: false` turns a hybrid
|
|
16
|
+
* reasoning model into a plain instruct model for the request.
|
|
17
|
+
*/
|
|
18
|
+
export class OpenAIChatTemplateKwargs extends Schema.Class<OpenAIChatTemplateKwargs>(
|
|
19
|
+
"OpenAIChatTemplateKwargs"
|
|
20
|
+
)({
|
|
21
|
+
enable_thinking: Schema.optionalKey(Schema.Boolean)
|
|
10
22
|
}) {}
|
|
11
23
|
|
|
12
24
|
export class OpenAIResponseFormat extends Schema.Class<OpenAIResponseFormat>(
|
|
@@ -30,7 +42,10 @@ export class OpenAIChatCompletionRequest extends Schema.Class<OpenAIChatCompleti
|
|
|
30
42
|
max_tokens: Schema.optionalKey(Schema.Int),
|
|
31
43
|
max_completion_tokens: Schema.optionalKey(Schema.Int),
|
|
32
44
|
stream: Schema.optionalKey(Schema.Boolean),
|
|
33
|
-
response_format: Schema.optionalKey(OpenAIResponseFormat)
|
|
45
|
+
response_format: Schema.optionalKey(OpenAIResponseFormat),
|
|
46
|
+
chat_template_kwargs: Schema.optionalKey(OpenAIChatTemplateKwargs),
|
|
47
|
+
logprobs: Schema.optionalKey(Schema.Boolean),
|
|
48
|
+
top_logprobs: Schema.optionalKey(Schema.Int)
|
|
34
49
|
}) {}
|
|
35
50
|
|
|
36
51
|
export class OpenAITokenUsage extends Schema.Class<OpenAITokenUsage>("OpenAITokenUsage")({
|
|
@@ -43,13 +58,35 @@ export class OpenAIChatResponseMessage extends Schema.Class<OpenAIChatResponseMe
|
|
|
43
58
|
"OpenAIChatResponseMessage"
|
|
44
59
|
)({
|
|
45
60
|
role: Schema.String,
|
|
46
|
-
content: Schema.NullOr(Schema.String)
|
|
61
|
+
content: Schema.NullOr(Schema.String),
|
|
62
|
+
// Local servers that split a reasoning model's output put the visible
|
|
63
|
+
// reply here when their parser mistakes it for thinking.
|
|
64
|
+
reasoning_content: Schema.optionalKey(Schema.NullOr(Schema.String))
|
|
65
|
+
}) {}
|
|
66
|
+
|
|
67
|
+
/** One candidate token at a generated position, with its log-probability. */
|
|
68
|
+
export class OpenAITopLogprob extends Schema.Class<OpenAITopLogprob>("OpenAITopLogprob")({
|
|
69
|
+
token: Schema.String,
|
|
70
|
+
logprob: Schema.Number
|
|
71
|
+
}) {}
|
|
72
|
+
|
|
73
|
+
export class OpenAITokenLogprob extends Schema.Class<OpenAITokenLogprob>("OpenAITokenLogprob")({
|
|
74
|
+
token: Schema.String,
|
|
75
|
+
logprob: Schema.Number,
|
|
76
|
+
top_logprobs: Schema.optionalKey(Schema.Array(OpenAITopLogprob))
|
|
77
|
+
}) {}
|
|
78
|
+
|
|
79
|
+
export class OpenAIChoiceLogprobs extends Schema.Class<OpenAIChoiceLogprobs>(
|
|
80
|
+
"OpenAIChoiceLogprobs"
|
|
81
|
+
)({
|
|
82
|
+
content: Schema.optionalKey(Schema.NullOr(Schema.Array(OpenAITokenLogprob)))
|
|
47
83
|
}) {}
|
|
48
84
|
|
|
49
85
|
export class OpenAIChatChoice extends Schema.Class<OpenAIChatChoice>("OpenAIChatChoice")({
|
|
50
86
|
index: Schema.Int.pipe(Schema.withConstructorDefault(Effect.succeed(0))),
|
|
51
87
|
message: Schema.optionalKey(OpenAIChatResponseMessage),
|
|
52
88
|
text: Schema.optionalKey(Schema.NullOr(Schema.String)),
|
|
89
|
+
logprobs: Schema.optionalKey(Schema.NullOr(OpenAIChoiceLogprobs)),
|
|
53
90
|
finish_reason: Schema.optionalKey(Schema.NullOr(Schema.String))
|
|
54
91
|
}) {}
|
|
55
92
|
|