@llm4ts/core 2.4.2 → 2.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/Connector.d.ts +4 -0
- package/dist/Connector.d.ts.map +1 -1
- package/dist/Connector.js +8 -3
- package/dist/Connector.js.map +1 -1
- package/dist/ConnectorConfig.d.ts.map +1 -1
- package/dist/ConnectorConfig.js +2 -0
- package/dist/ConnectorConfig.js.map +1 -1
- package/dist/ContextManagement.d.ts.map +1 -1
- package/dist/ContextManagement.js +1 -0
- package/dist/ContextManagement.js.map +1 -1
- package/dist/LabelScoring.d.ts +52 -0
- package/dist/LabelScoring.d.ts.map +1 -0
- package/dist/LabelScoring.js +139 -0
- package/dist/LabelScoring.js.map +1 -0
- package/dist/LlmService.d.ts +30 -2
- package/dist/LlmService.d.ts.map +1 -1
- package/dist/LlmService.js.map +1 -1
- package/dist/Models.d.ts +34 -2
- package/dist/Models.d.ts.map +1 -1
- package/dist/Models.js +37 -1
- package/dist/Models.js.map +1 -1
- package/dist/eval/Judge.d.ts +15 -0
- package/dist/eval/Judge.d.ts.map +1 -1
- package/dist/eval/Judge.js +50 -0
- package/dist/eval/Judge.js.map +1 -1
- package/dist/judgment/FakeJudgment.d.ts +22 -0
- package/dist/judgment/FakeJudgment.d.ts.map +1 -0
- package/dist/judgment/FakeJudgment.js +42 -0
- package/dist/judgment/FakeJudgment.js.map +1 -0
- package/dist/judgment/Judgment.d.ts +57 -0
- package/dist/judgment/Judgment.d.ts.map +1 -0
- package/dist/judgment/Judgment.js +57 -0
- package/dist/judgment/Judgment.js.map +1 -0
- package/dist/judgment/LlmJudgment.d.ts +73 -0
- package/dist/judgment/LlmJudgment.d.ts.map +1 -0
- package/dist/judgment/LlmJudgment.js +239 -0
- package/dist/judgment/LlmJudgment.js.map +1 -0
- package/dist/judgment/Schemas.d.ts +163 -0
- package/dist/judgment/Schemas.d.ts.map +1 -0
- package/dist/judgment/Schemas.js +197 -0
- package/dist/judgment/Schemas.js.map +1 -0
- package/dist/judgment/TypeSafeJudgment.d.ts +46 -0
- package/dist/judgment/TypeSafeJudgment.d.ts.map +1 -0
- package/dist/judgment/TypeSafeJudgment.js +185 -0
- package/dist/judgment/TypeSafeJudgment.js.map +1 -0
- package/dist/observability/MeteredLlmService.d.ts.map +1 -1
- package/dist/observability/MeteredLlmService.js +1 -0
- package/dist/observability/MeteredLlmService.js.map +1 -1
- package/dist/providers/ConnectorFactories.d.ts.map +1 -1
- package/dist/providers/ConnectorFactories.js +2 -0
- package/dist/providers/ConnectorFactories.js.map +1 -1
- package/dist/providers/LmStudioProvider.d.ts.map +1 -1
- package/dist/providers/LmStudioProvider.js +45 -34
- package/dist/providers/LmStudioProvider.js.map +1 -1
- package/dist/providers/MlxLmProvider.d.ts +29 -0
- package/dist/providers/MlxLmProvider.d.ts.map +1 -0
- package/dist/providers/MlxLmProvider.js +249 -0
- package/dist/providers/MlxLmProvider.js.map +1 -0
- package/dist/providers/MockProvider.d.ts.map +1 -1
- package/dist/providers/MockProvider.js +11 -0
- package/dist/providers/MockProvider.js.map +1 -1
- package/dist/providers/OpenAIModels.d.ts +35 -0
- package/dist/providers/OpenAIModels.d.ts.map +1 -1
- package/dist/providers/OpenAIModels.js +36 -3
- package/dist/providers/OpenAIModels.js.map +1 -1
- package/package.json +8 -1
- package/src/Connector.ts +13 -3
- package/src/ConnectorConfig.ts +2 -0
- package/src/ContextManagement.ts +1 -0
- package/src/LabelScoring.ts +211 -0
- package/src/LlmService.ts +37 -1
- package/src/Models.ts +43 -1
- package/src/eval/Judge.ts +74 -0
- package/src/judgment/FakeJudgment.ts +90 -0
- package/src/judgment/Judgment.ts +123 -0
- package/src/judgment/LlmJudgment.ts +399 -0
- package/src/judgment/Schemas.ts +278 -0
- package/src/judgment/TypeSafeJudgment.ts +253 -0
- package/src/observability/MeteredLlmService.ts +7 -0
- package/src/providers/ConnectorFactories.ts +4 -0
- package/src/providers/LmStudioProvider.ts +64 -48
- package/src/providers/MlxLmProvider.ts +357 -0
- package/src/providers/MockProvider.ts +19 -0
- package/src/providers/OpenAIModels.ts +40 -3
|
@@ -0,0 +1,211 @@
|
|
|
1
|
+
import * as Effect from "effect/Effect"
|
|
2
|
+
import * as Schema from "effect/Schema"
|
|
3
|
+
import { InvalidRequestError, ParseError, type LlmError } from "./Errors.ts"
|
|
4
|
+
import * as Result from "effect/Result"
|
|
5
|
+
import type { LabelSequence, LlmServiceShape } from "./LlmService.ts"
|
|
6
|
+
import { LabelDistribution, type JsonSchema, type LabelMethod } from "./Models.ts"
|
|
7
|
+
|
|
8
|
+
/**
|
|
9
|
+
* Label scoring: the one classification primitive every connector offers.
|
|
10
|
+
*
|
|
11
|
+
* A caller hands over a prompt and a closed set of labels and gets a
|
|
12
|
+
* probability per label back. Connectors that expose token log-probabilities
|
|
13
|
+
* implement it natively in one forward pass; everything else derives it here
|
|
14
|
+
* from `executeStructured`, asking the model to write the numbers down
|
|
15
|
+
* (method `verbalized`, which callers should hold to a higher bar).
|
|
16
|
+
*/
|
|
17
|
+
|
|
18
|
+
const raw = (text: string): string => (text.length > 200 ? `${text.slice(0, 200)}…` : text)
|
|
19
|
+
|
|
20
|
+
/**
|
|
21
|
+
* Keep only the offered labels, clamp negatives to zero and renormalize.
|
|
22
|
+
* `support` records how much mass the offered labels held before
|
|
23
|
+
* renormalization: for log-probabilities that is their absolute share of
|
|
24
|
+
* the vocabulary distribution (so one label found with 5% of the mass
|
|
25
|
+
* yields probability 1 but support 0.05); a backend whose numbers were
|
|
26
|
+
* declared over the labels alone passes `support: 1`. Fails typed when
|
|
27
|
+
* none of the offered labels carries any mass, so a caller never branches
|
|
28
|
+
* on a silent uniform distribution.
|
|
29
|
+
*/
|
|
30
|
+
export const normalizeLabelProbabilities = (
|
|
31
|
+
labels: ReadonlyArray<string>,
|
|
32
|
+
observed: Readonly<Record<string, number>>,
|
|
33
|
+
method: LabelMethod,
|
|
34
|
+
extra: {
|
|
35
|
+
readonly usage?: LabelDistribution["usage"]
|
|
36
|
+
readonly model?: string
|
|
37
|
+
readonly support?: number
|
|
38
|
+
} = {}
|
|
39
|
+
): Effect.Effect<LabelDistribution, ParseError> => {
|
|
40
|
+
const kept = labels.map((label) => [label, Math.max(0, observed[label] ?? 0)] as const)
|
|
41
|
+
const total = kept.reduce((sum, [, value]) => sum + value, 0)
|
|
42
|
+
if (!(total > 0)) {
|
|
43
|
+
return Effect.fail(
|
|
44
|
+
ParseError.make({
|
|
45
|
+
message: `none of the offered labels was observed: ${labels.join(", ")}`,
|
|
46
|
+
raw: raw(JSON.stringify(observed))
|
|
47
|
+
})
|
|
48
|
+
)
|
|
49
|
+
}
|
|
50
|
+
return Effect.succeed(
|
|
51
|
+
LabelDistribution.make({
|
|
52
|
+
probabilities: Object.fromEntries(kept.map(([label, value]) => [label, value / total])),
|
|
53
|
+
method,
|
|
54
|
+
support: Math.max(0, Math.min(1, extra.support ?? total)),
|
|
55
|
+
...(extra.usage === undefined ? {} : { usage: extra.usage }),
|
|
56
|
+
...(extra.model === undefined ? {} : { model: extra.model })
|
|
57
|
+
})
|
|
58
|
+
)
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
export class VerbalizedLabels extends Schema.Class<VerbalizedLabels>("VerbalizedLabels")({
|
|
62
|
+
label: Schema.String,
|
|
63
|
+
probabilities: Schema.Record(Schema.String, Schema.Number)
|
|
64
|
+
}) {}
|
|
65
|
+
|
|
66
|
+
export const verbalizedLabelsJsonSchema = (labels: ReadonlyArray<string>): JsonSchema => ({
|
|
67
|
+
type: "object",
|
|
68
|
+
properties: {
|
|
69
|
+
label: { type: "string", enum: [...labels] },
|
|
70
|
+
probabilities: {
|
|
71
|
+
type: "object",
|
|
72
|
+
properties: Object.fromEntries(labels.map((label) => [label, { type: "number" }])),
|
|
73
|
+
required: [...labels],
|
|
74
|
+
additionalProperties: false
|
|
75
|
+
}
|
|
76
|
+
},
|
|
77
|
+
required: ["label", "probabilities"],
|
|
78
|
+
additionalProperties: false
|
|
79
|
+
})
|
|
80
|
+
|
|
81
|
+
export const verbalizedLabelsPrompt = (prompt: string, labels: ReadonlyArray<string>): string =>
|
|
82
|
+
`${prompt}\n\nReply with JSON only: {"label": <the one label you choose>, "probabilities": {<a probability for every label, summing to 1>}}. Labels: ${labels.join(", ")}.`
|
|
83
|
+
|
|
84
|
+
/**
|
|
85
|
+
* The default `scoreLabels` for a connector without log-probabilities: one
|
|
86
|
+
* schema-constrained JSON reply, renormalized over the offered labels.
|
|
87
|
+
*/
|
|
88
|
+
export const verbalizedScoreLabels =
|
|
89
|
+
(executeStructuredWithUsage: LlmServiceShape["executeStructuredWithUsage"]) =>
|
|
90
|
+
(prompt: string, labels: ReadonlyArray<string>): Effect.Effect<LabelDistribution, LlmError> =>
|
|
91
|
+
executeStructuredWithUsage(
|
|
92
|
+
verbalizedLabelsPrompt(prompt, labels),
|
|
93
|
+
VerbalizedLabels,
|
|
94
|
+
verbalizedLabelsJsonSchema(labels)
|
|
95
|
+
).pipe(
|
|
96
|
+
Effect.flatMap(([reply, usage, model]) =>
|
|
97
|
+
// A reply that names a label but gives it no mass contradicts itself;
|
|
98
|
+
// failing here is what lets the judgment layer retry or hold, instead
|
|
99
|
+
// of a manufactured 1.0 that could wave a review through.
|
|
100
|
+
labels.includes(reply.label) && !((reply.probabilities[reply.label] ?? 0) > 0)
|
|
101
|
+
? Effect.fail(
|
|
102
|
+
ParseError.make({
|
|
103
|
+
message: `the model chose "${reply.label}" but gave it no probability`,
|
|
104
|
+
raw: raw(JSON.stringify(reply.probabilities))
|
|
105
|
+
})
|
|
106
|
+
)
|
|
107
|
+
: normalizeLabelProbabilities(labels, reply.probabilities, "verbalized", {
|
|
108
|
+
// Declared over the labels alone: support is by construction.
|
|
109
|
+
support: 1,
|
|
110
|
+
...(usage === undefined ? {} : { usage }),
|
|
111
|
+
...(model === undefined ? {} : { model })
|
|
112
|
+
})
|
|
113
|
+
)
|
|
114
|
+
)
|
|
115
|
+
|
|
116
|
+
/** For fakes and adapters that cannot classify: fails typed instead of guessing. */
|
|
117
|
+
export const unsupportedScoreLabels: LlmServiceShape["scoreLabels"] = (_prompt, _labels) =>
|
|
118
|
+
Effect.fail(InvalidRequestError.make({ message: "label scoring is not supported here" }))
|
|
119
|
+
|
|
120
|
+
export class VerbalizedLabelSequence extends Schema.Class<VerbalizedLabelSequence>(
|
|
121
|
+
"VerbalizedLabelSequence"
|
|
122
|
+
)({
|
|
123
|
+
answers: Schema.Array(VerbalizedLabels)
|
|
124
|
+
}) {}
|
|
125
|
+
|
|
126
|
+
export const verbalizedLabelSequenceJsonSchema = (
|
|
127
|
+
labelSets: ReadonlyArray<ReadonlyArray<string>>
|
|
128
|
+
): JsonSchema => ({
|
|
129
|
+
type: "object",
|
|
130
|
+
properties: {
|
|
131
|
+
answers: {
|
|
132
|
+
type: "array",
|
|
133
|
+
minItems: labelSets.length,
|
|
134
|
+
maxItems: labelSets.length,
|
|
135
|
+
items: {
|
|
136
|
+
type: "object",
|
|
137
|
+
properties: {
|
|
138
|
+
label: { type: "string" },
|
|
139
|
+
probabilities: { type: "object", additionalProperties: { type: "number" } }
|
|
140
|
+
},
|
|
141
|
+
required: ["label", "probabilities"]
|
|
142
|
+
}
|
|
143
|
+
}
|
|
144
|
+
},
|
|
145
|
+
required: ["answers"],
|
|
146
|
+
additionalProperties: false
|
|
147
|
+
})
|
|
148
|
+
|
|
149
|
+
export const verbalizedLabelSequencePrompt = (
|
|
150
|
+
prompt: string,
|
|
151
|
+
labelSets: ReadonlyArray<ReadonlyArray<string>>
|
|
152
|
+
): string =>
|
|
153
|
+
`${prompt}\n\nReply with JSON only: {"answers": [<one entry per question, in order>]}, each entry {"label": <the one label you choose>, "probabilities": {<a probability for every label of that question, summing to 1>}}. Labels per question: ${labelSets
|
|
154
|
+
.map((labels, index) => `${index + 1}: ${labels.join(", ")}`)
|
|
155
|
+
.join("; ")}.`
|
|
156
|
+
|
|
157
|
+
/**
|
|
158
|
+
* The derived `scoreLabelSequence`: one schema-constrained JSON reply with
|
|
159
|
+
* an answer per question. A missing position, or a chosen label with no
|
|
160
|
+
* mass, fails that position only.
|
|
161
|
+
*/
|
|
162
|
+
export const verbalizedScoreLabelSequence =
|
|
163
|
+
(executeStructuredWithUsage: LlmServiceShape["executeStructuredWithUsage"]) =>
|
|
164
|
+
(
|
|
165
|
+
prompt: string,
|
|
166
|
+
labelSets: ReadonlyArray<ReadonlyArray<string>>
|
|
167
|
+
): Effect.Effect<LabelSequence, LlmError> =>
|
|
168
|
+
executeStructuredWithUsage(
|
|
169
|
+
verbalizedLabelSequencePrompt(prompt, labelSets),
|
|
170
|
+
VerbalizedLabelSequence,
|
|
171
|
+
verbalizedLabelSequenceJsonSchema(labelSets)
|
|
172
|
+
).pipe(
|
|
173
|
+
Effect.flatMap(([reply, usage, model]) =>
|
|
174
|
+
Effect.forEach(labelSets, (labels, index) => {
|
|
175
|
+
const answer = reply.answers[index]
|
|
176
|
+
const entry: Effect.Effect<LabelDistribution, ParseError> =
|
|
177
|
+
answer === undefined
|
|
178
|
+
? Effect.fail(
|
|
179
|
+
ParseError.make({
|
|
180
|
+
message: `no answer for question ${index + 1} of ${labelSets.length}`,
|
|
181
|
+
raw: raw(JSON.stringify(reply.answers))
|
|
182
|
+
})
|
|
183
|
+
)
|
|
184
|
+
: labels.includes(answer.label) && !((answer.probabilities[answer.label] ?? 0) > 0)
|
|
185
|
+
? Effect.fail(
|
|
186
|
+
ParseError.make({
|
|
187
|
+
message: `question ${index + 1}: the model chose "${answer.label}" but gave it no probability`,
|
|
188
|
+
raw: raw(JSON.stringify(answer.probabilities))
|
|
189
|
+
})
|
|
190
|
+
)
|
|
191
|
+
: normalizeLabelProbabilities(labels, answer.probabilities, "verbalized", {
|
|
192
|
+
support: 1
|
|
193
|
+
})
|
|
194
|
+
return Effect.result(entry)
|
|
195
|
+
}).pipe(
|
|
196
|
+
Effect.map(
|
|
197
|
+
(entries): LabelSequence => ({
|
|
198
|
+
entries,
|
|
199
|
+
...(usage === undefined ? {} : { usage }),
|
|
200
|
+
...(model === undefined ? {} : { model })
|
|
201
|
+
})
|
|
202
|
+
)
|
|
203
|
+
)
|
|
204
|
+
)
|
|
205
|
+
)
|
|
206
|
+
|
|
207
|
+
/** Convenience for callers that want the sequence's successes only. */
|
|
208
|
+
export const sequenceSuccesses = (
|
|
209
|
+
sequence: LabelSequence
|
|
210
|
+
): ReadonlyArray<LabelDistribution | undefined> =>
|
|
211
|
+
sequence.entries.map((entry) => (Result.isSuccess(entry) ? entry.success : undefined))
|
package/src/LlmService.ts
CHANGED
|
@@ -2,9 +2,11 @@ import * as Context from "effect/Context"
|
|
|
2
2
|
import type * as Effect from "effect/Effect"
|
|
3
3
|
import type * as Stream from "effect/Stream"
|
|
4
4
|
import type * as Schema from "effect/Schema"
|
|
5
|
-
import type
|
|
5
|
+
import type * as Result from "effect/Result"
|
|
6
|
+
import type { LlmError, ParseError } from "./Errors.ts"
|
|
6
7
|
import type {
|
|
7
8
|
JsonSchema,
|
|
9
|
+
LabelDistribution,
|
|
8
10
|
LlmChunk,
|
|
9
11
|
Message,
|
|
10
12
|
TokenUsage,
|
|
@@ -37,9 +39,43 @@ export interface LlmServiceShape {
|
|
|
37
39
|
schema: Schema.ConstraintCodec<A, E, RD, RE>,
|
|
38
40
|
jsonSchema: JsonSchema
|
|
39
41
|
) => Effect.Effect<StructuredResult<A>, LlmError, RD>
|
|
42
|
+
/**
|
|
43
|
+
* One atomic classification: the probability of each offered label given
|
|
44
|
+
* the prompt. Connectors with token log-probabilities answer in one forward
|
|
45
|
+
* pass; the rest derive it from `executeStructured` (see `LabelScoring`).
|
|
46
|
+
* Fails with `ParseError` when no offered label could be observed.
|
|
47
|
+
*/
|
|
48
|
+
readonly scoreLabels: (
|
|
49
|
+
prompt: string,
|
|
50
|
+
labels: ReadonlyArray<string>
|
|
51
|
+
) => Effect.Effect<LabelDistribution, LlmError>
|
|
52
|
+
/**
|
|
53
|
+
* Several label questions answered in one call over a shared prompt
|
|
54
|
+
* prefix (the judgment layer's `shared-prefix` batching): the reply is one
|
|
55
|
+
* label per question in order, and each position is read as its own
|
|
56
|
+
* distribution. Optional, unlike `scoreLabels`: it is a cost optimization
|
|
57
|
+
* that only backends with token log-probabilities implement natively, and
|
|
58
|
+
* the judgment layer derives a verbalized version from structured output
|
|
59
|
+
* when it is absent, so no fake or decorator has to carry it.
|
|
60
|
+
*/
|
|
61
|
+
readonly scoreLabelSequence?: (
|
|
62
|
+
prompt: string,
|
|
63
|
+
labelSets: ReadonlyArray<ReadonlyArray<string>>
|
|
64
|
+
) => Effect.Effect<LabelSequence, LlmError>
|
|
40
65
|
readonly isAvailable: Effect.Effect<boolean>
|
|
41
66
|
}
|
|
42
67
|
|
|
68
|
+
/**
|
|
69
|
+
* The reply to `scoreLabelSequence`: one outcome per question in order (a
|
|
70
|
+
* position that could not be read fails on its own, so the caller can fall
|
|
71
|
+
* back for that question only), plus the call's usage and model once.
|
|
72
|
+
*/
|
|
73
|
+
export interface LabelSequence {
|
|
74
|
+
readonly entries: ReadonlyArray<Result.Result<LabelDistribution, ParseError>>
|
|
75
|
+
readonly usage?: TokenUsage
|
|
76
|
+
readonly model?: string
|
|
77
|
+
}
|
|
78
|
+
|
|
43
79
|
export class LlmService extends Context.Service<LlmService, LlmServiceShape>()(
|
|
44
80
|
"@llm4ts/core/LlmService"
|
|
45
81
|
) {}
|
package/src/Models.ts
CHANGED
|
@@ -9,6 +9,7 @@ export const LlmProvider = Schema.Literals([
|
|
|
9
9
|
"Anthropic",
|
|
10
10
|
"LmStudio",
|
|
11
11
|
"Ollama",
|
|
12
|
+
"MlxLm",
|
|
12
13
|
"OpenCode",
|
|
13
14
|
"Mock"
|
|
14
15
|
])
|
|
@@ -24,6 +25,7 @@ export const ConnectorIds = Object.freeze({
|
|
|
24
25
|
GeminiApi: new ConnectorId({ value: "gemini-api" }),
|
|
25
26
|
LmStudio: new ConnectorId({ value: "lm-studio" }),
|
|
26
27
|
Ollama: new ConnectorId({ value: "ollama" }),
|
|
28
|
+
MlxLm: new ConnectorId({ value: "mlx-lm" }),
|
|
27
29
|
ClaudeCli: new ConnectorId({ value: "claude-cli" }),
|
|
28
30
|
GeminiCli: new ConnectorId({ value: "gemini-cli" }),
|
|
29
31
|
OpenCode: new ConnectorId({ value: "opencode" }),
|
|
@@ -41,7 +43,8 @@ export const apiConnectorIds: ReadonlyArray<ConnectorId> = Object.freeze([
|
|
|
41
43
|
ConnectorIds.Anthropic,
|
|
42
44
|
ConnectorIds.GeminiApi,
|
|
43
45
|
ConnectorIds.LmStudio,
|
|
44
|
-
ConnectorIds.Ollama
|
|
46
|
+
ConnectorIds.Ollama,
|
|
47
|
+
ConnectorIds.MlxLm
|
|
45
48
|
])
|
|
46
49
|
|
|
47
50
|
export const cliConnectorIds: ReadonlyArray<ConnectorId> = Object.freeze([
|
|
@@ -77,6 +80,8 @@ export const defaultBaseUrl = (provider: LlmProvider): string | undefined => {
|
|
|
77
80
|
return "http://localhost:1234/v1"
|
|
78
81
|
case "Ollama":
|
|
79
82
|
return "http://localhost:11434"
|
|
83
|
+
case "MlxLm":
|
|
84
|
+
return "http://localhost:8080"
|
|
80
85
|
case "OpenCode":
|
|
81
86
|
return "http://localhost:4096"
|
|
82
87
|
}
|
|
@@ -88,6 +93,7 @@ const connectorProviderTable: Readonly<Record<string, LlmProvider>> = {
|
|
|
88
93
|
"gemini-api": "GeminiApi",
|
|
89
94
|
"lm-studio": "LmStudio",
|
|
90
95
|
ollama: "Ollama",
|
|
96
|
+
"mlx-lm": "MlxLm",
|
|
91
97
|
opencode: "OpenCode",
|
|
92
98
|
"gemini-cli": "GeminiCli",
|
|
93
99
|
mock: "Mock"
|
|
@@ -115,6 +121,8 @@ export const providerConnectorId = (provider: LlmProvider): ConnectorId => {
|
|
|
115
121
|
return ConnectorIds.LmStudio
|
|
116
122
|
case "Ollama":
|
|
117
123
|
return ConnectorIds.Ollama
|
|
124
|
+
case "MlxLm":
|
|
125
|
+
return ConnectorIds.MlxLm
|
|
118
126
|
case "OpenCode":
|
|
119
127
|
return ConnectorIds.OpenCode
|
|
120
128
|
case "Mock":
|
|
@@ -255,6 +263,37 @@ export type InteractionSupport = typeof InteractionSupport.Type
|
|
|
255
263
|
export const ReadOnlyEnforcement = Schema.Literals(["enforced", "advisory", "ignored"])
|
|
256
264
|
export type ReadOnlyEnforcement = typeof ReadOnlyEnforcement.Type
|
|
257
265
|
|
|
266
|
+
/**
|
|
267
|
+
* How a connector produces the per-label probabilities behind `scoreLabels`:
|
|
268
|
+
* - "logprobs": read off the backend's token log-probabilities in one
|
|
269
|
+
* forward pass (a real distribution, still uncalibrated).
|
|
270
|
+
* - "verbalized": the model writes the numbers itself in a JSON reply
|
|
271
|
+
* (typically overconfident; hold to a higher bar).
|
|
272
|
+
* - "none": the connector cannot answer label questions at all.
|
|
273
|
+
*/
|
|
274
|
+
export const LabelProbabilities = Schema.Literals(["logprobs", "verbalized", "none"])
|
|
275
|
+
export type LabelProbabilities = typeof LabelProbabilities.Type
|
|
276
|
+
|
|
277
|
+
/** How a label distribution's numbers were extracted (see `LabelProbabilities`). */
|
|
278
|
+
export const LabelMethod = Schema.Literals(["logprobs", "verbalized", "sampled"])
|
|
279
|
+
export type LabelMethod = typeof LabelMethod.Type
|
|
280
|
+
|
|
281
|
+
/**
|
|
282
|
+
* A probability per label, normalized over the labels the caller offered.
|
|
283
|
+
* `support` is the probability mass the backend actually placed on those
|
|
284
|
+
* labels before renormalization (1 when the numbers were declared over the
|
|
285
|
+
* labels alone): a distribution renormalized from a sliver of mass is not
|
|
286
|
+
* a confident one, and callers must not treat it as one.
|
|
287
|
+
* `usage` is whatever the backend reported for the call, if anything.
|
|
288
|
+
*/
|
|
289
|
+
export class LabelDistribution extends Schema.Class<LabelDistribution>("LabelDistribution")({
|
|
290
|
+
probabilities: Schema.Record(Schema.String, Schema.Number),
|
|
291
|
+
method: LabelMethod,
|
|
292
|
+
support: Schema.Number,
|
|
293
|
+
usage: Schema.optionalKey(TokenUsage),
|
|
294
|
+
model: Schema.optionalKey(Schema.String)
|
|
295
|
+
}) {}
|
|
296
|
+
|
|
258
297
|
export class ConnectorCapabilities extends Schema.Class<ConnectorCapabilities>(
|
|
259
298
|
"ConnectorCapabilities"
|
|
260
299
|
)({
|
|
@@ -265,6 +304,9 @@ export class ConnectorCapabilities extends Schema.Class<ConnectorCapabilities>(
|
|
|
265
304
|
approval: Schema.Boolean.pipe(Schema.withConstructorDefault(Effect.succeed(false))),
|
|
266
305
|
structuredOutput: Schema.Boolean.pipe(Schema.withConstructorDefault(Effect.succeed(true))),
|
|
267
306
|
usageReporting: Schema.Boolean.pipe(Schema.withConstructorDefault(Effect.succeed(true))),
|
|
307
|
+
labelProbabilities: LabelProbabilities.pipe(
|
|
308
|
+
Schema.withConstructorDefault(Effect.succeed<LabelProbabilities>("verbalized"))
|
|
309
|
+
),
|
|
268
310
|
readOnlyEnforcement: ReadOnlyEnforcement.pipe(
|
|
269
311
|
Schema.withConstructorDefault(Effect.succeed<ReadOnlyEnforcement>("advisory"))
|
|
270
312
|
)
|
package/src/eval/Judge.ts
CHANGED
|
@@ -1,5 +1,8 @@
|
|
|
1
1
|
import * as Effect from "effect/Effect"
|
|
2
2
|
import * as Schema from "effect/Schema"
|
|
3
|
+
import { ProviderError } from "../Errors.ts"
|
|
4
|
+
import type { JudgmentShape } from "../judgment/Judgment.ts"
|
|
5
|
+
import { score, type ScoreQuestion, type State } from "../judgment/Schemas.ts"
|
|
3
6
|
import type { LlmServiceShape } from "../LlmService.ts"
|
|
4
7
|
import type { JsonSchema } from "../Models.ts"
|
|
5
8
|
import { DimensionScore, EvalResult } from "./Eval.ts"
|
|
@@ -90,3 +93,74 @@ export const judge = (
|
|
|
90
93
|
.executeStructured(buildPrompt(system, dimensions, sample), JudgeResponse, judgeJsonSchema)
|
|
91
94
|
.pipe(Effect.map((response) => toResult(dimensions, response)))
|
|
92
95
|
)
|
|
96
|
+
|
|
97
|
+
/**
|
|
98
|
+
* A dimension as a Score question: one level per rubric point, so the
|
|
99
|
+
* judgment returns a distribution over the scale instead of one integer.
|
|
100
|
+
*/
|
|
101
|
+
export const dimensionQuestion = (dimension: Dimension): ScoreQuestion =>
|
|
102
|
+
score(
|
|
103
|
+
`${dimension.name}: ${dimension.rubric}`,
|
|
104
|
+
Array.from({ length: dimension.maxScore + 1 }, (_, level) => ({
|
|
105
|
+
level,
|
|
106
|
+
of: dimension.maxScore,
|
|
107
|
+
meaning:
|
|
108
|
+
level === 0
|
|
109
|
+
? "does not meet the rubric at all"
|
|
110
|
+
: level === dimension.maxScore
|
|
111
|
+
? "fully meets the rubric"
|
|
112
|
+
: `partially meets the rubric (${level} of ${dimension.maxScore})`
|
|
113
|
+
}))
|
|
114
|
+
)
|
|
115
|
+
|
|
116
|
+
export const sampleState = (sample: Sample): State => ({
|
|
117
|
+
...(sample.query === undefined ? {} : { query: sample.query }),
|
|
118
|
+
...(sample.context === undefined ? {} : { context: sample.context }),
|
|
119
|
+
response: sample.response,
|
|
120
|
+
...(sample.expected === undefined ? {} : { expected: sample.expected })
|
|
121
|
+
})
|
|
122
|
+
|
|
123
|
+
/**
|
|
124
|
+
* The rubric judge over the Judgment service (ADR 0017, consumer 1): every
|
|
125
|
+
* dimension is an independent Score question over the same sample. A
|
|
126
|
+
* dimension the backend could not answer scores 0 with the reason recorded,
|
|
127
|
+
* matching the generative judge's treatment of a missing score.
|
|
128
|
+
*/
|
|
129
|
+
export const judgeWithJudgment = (
|
|
130
|
+
judgment: JudgmentShape,
|
|
131
|
+
dimensions: ReadonlyArray<Dimension>
|
|
132
|
+
): Evaluator<Sample> =>
|
|
133
|
+
makeEvaluator((sample) =>
|
|
134
|
+
judgment
|
|
135
|
+
.judge({
|
|
136
|
+
state: sampleState(sample),
|
|
137
|
+
questions: Object.fromEntries(
|
|
138
|
+
dimensions.map((dimension) => [dimension.name, dimensionQuestion(dimension)])
|
|
139
|
+
)
|
|
140
|
+
})
|
|
141
|
+
.pipe(
|
|
142
|
+
Effect.mapError((error) =>
|
|
143
|
+
ProviderError.make({ message: `judgment backend failed: ${error.message}`, cause: error })
|
|
144
|
+
),
|
|
145
|
+
Effect.map((result) =>
|
|
146
|
+
EvalResult.make({
|
|
147
|
+
scores: dimensions.map((dimension) => {
|
|
148
|
+
const answer = result.answers[dimension.name]
|
|
149
|
+
if (answer === undefined || answer.type !== "score") {
|
|
150
|
+
const failure = result.failures.find((entry) => entry.key === dimension.name)
|
|
151
|
+
return DimensionScore.make({
|
|
152
|
+
name: dimension.name,
|
|
153
|
+
score: 0,
|
|
154
|
+
reasoning: failure === undefined ? "missing" : `failed: ${failure.reason}`
|
|
155
|
+
})
|
|
156
|
+
}
|
|
157
|
+
return DimensionScore.make({
|
|
158
|
+
name: dimension.name,
|
|
159
|
+
score: Math.max(0, Math.min(dimension.maxScore, Math.round(answer.score))),
|
|
160
|
+
reasoning: `${answer.origin.method} judgment (${answer.origin.backend}), confidence ${answer.confidence.toFixed(2)}, support ${answer.support.toFixed(2)}, expected level ${answer.score.toFixed(2)}`
|
|
161
|
+
})
|
|
162
|
+
})
|
|
163
|
+
})
|
|
164
|
+
)
|
|
165
|
+
)
|
|
166
|
+
)
|
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
import * as Effect from "effect/Effect"
|
|
2
|
+
import * as Layer from "effect/Layer"
|
|
3
|
+
import * as Ref from "effect/Ref"
|
|
4
|
+
import { Judgment, type JudgmentInput, type JudgmentShape } from "./Judgment.ts"
|
|
5
|
+
import {
|
|
6
|
+
choiceAnswer,
|
|
7
|
+
JudgmentRequest,
|
|
8
|
+
JudgmentResult,
|
|
9
|
+
origins,
|
|
10
|
+
QuestionFailure,
|
|
11
|
+
scoreAnswer,
|
|
12
|
+
truthAnswer,
|
|
13
|
+
type Answer,
|
|
14
|
+
type Question
|
|
15
|
+
} from "./Schemas.ts"
|
|
16
|
+
|
|
17
|
+
/**
|
|
18
|
+
* `FakeJudgment`: deterministic answers for tests. A plan maps question keys
|
|
19
|
+
* to canned answers (or a failure reason); unplanned questions get a
|
|
20
|
+
* predictable default so a consumer test can run without listing every key.
|
|
21
|
+
*/
|
|
22
|
+
|
|
23
|
+
export interface FakeJudgmentPlan {
|
|
24
|
+
readonly answers?: Readonly<Record<string, Answer>>
|
|
25
|
+
readonly failures?: Readonly<Record<string, string>>
|
|
26
|
+
/** Truth probability for unplanned truth questions (default 1). */
|
|
27
|
+
readonly defaultTruth?: number
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
export interface FakeJudgment {
|
|
31
|
+
readonly judgment: JudgmentShape
|
|
32
|
+
readonly recorded: Effect.Effect<ReadonlyArray<JudgmentRequest>>
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
const defaultAnswer = (question: Question, defaultTruth: number): Answer => {
|
|
36
|
+
switch (question.type) {
|
|
37
|
+
case "choice": {
|
|
38
|
+
const keys = Object.keys(question.criteria)
|
|
39
|
+
return choiceAnswer(
|
|
40
|
+
Object.fromEntries(keys.map((key, index) => [key, index === 0 ? 1 : 0])),
|
|
41
|
+
origins.fake()
|
|
42
|
+
)
|
|
43
|
+
}
|
|
44
|
+
case "score":
|
|
45
|
+
return scoreAnswer(
|
|
46
|
+
question,
|
|
47
|
+
Object.fromEntries(
|
|
48
|
+
question.criteria.map((_, index) => [String(index), index === 0 ? 1 : 0])
|
|
49
|
+
),
|
|
50
|
+
origins.fake()
|
|
51
|
+
)
|
|
52
|
+
case "truth":
|
|
53
|
+
return truthAnswer(defaultTruth, origins.fake())
|
|
54
|
+
}
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
export const makeFakeJudgment = Effect.fn("@llm4ts/core/judgment/FakeJudgment.make")(function* (
|
|
58
|
+
plan: FakeJudgmentPlan = {}
|
|
59
|
+
): Effect.fn.Return<FakeJudgment> {
|
|
60
|
+
const requests = yield* Ref.make<ReadonlyArray<JudgmentRequest>>([])
|
|
61
|
+
const judge = (input: JudgmentInput): Effect.Effect<JudgmentResult> =>
|
|
62
|
+
Ref.update(requests, (all) => [
|
|
63
|
+
...all,
|
|
64
|
+
JudgmentRequest.make({ state: input.state, questions: input.questions })
|
|
65
|
+
]).pipe(
|
|
66
|
+
Effect.map(() => {
|
|
67
|
+
const answers: Record<string, Answer> = {}
|
|
68
|
+
const failures: Array<QuestionFailure> = []
|
|
69
|
+
for (const [key, question] of Object.entries(input.questions)) {
|
|
70
|
+
const failure = plan.failures?.[key]
|
|
71
|
+
if (failure !== undefined) {
|
|
72
|
+
failures.push(QuestionFailure.make({ key, reason: failure }))
|
|
73
|
+
continue
|
|
74
|
+
}
|
|
75
|
+
answers[key] = plan.answers?.[key] ?? defaultAnswer(question, plan.defaultTruth ?? 1)
|
|
76
|
+
}
|
|
77
|
+
return JudgmentResult.make({ answers, failures, backend: "fake" })
|
|
78
|
+
})
|
|
79
|
+
)
|
|
80
|
+
return {
|
|
81
|
+
judgment: { backend: "fake", identity: "fake", judge },
|
|
82
|
+
recorded: Ref.get(requests)
|
|
83
|
+
}
|
|
84
|
+
})
|
|
85
|
+
|
|
86
|
+
export const FakeJudgmentLive = (plan?: FakeJudgmentPlan): Layer.Layer<Judgment> =>
|
|
87
|
+
Layer.effect(
|
|
88
|
+
Judgment,
|
|
89
|
+
Effect.map(makeFakeJudgment(plan), (fake) => fake.judgment)
|
|
90
|
+
)
|
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
import * as Context from "effect/Context"
|
|
2
|
+
import * as Effect from "effect/Effect"
|
|
3
|
+
import * as Schema from "effect/Schema"
|
|
4
|
+
import {
|
|
5
|
+
ChoiceAnswer,
|
|
6
|
+
ScoreAnswer,
|
|
7
|
+
TruthAnswer,
|
|
8
|
+
type Answer,
|
|
9
|
+
type JudgmentBackend,
|
|
10
|
+
type JudgmentResult,
|
|
11
|
+
type Question,
|
|
12
|
+
type State
|
|
13
|
+
} from "./Schemas.ts"
|
|
14
|
+
|
|
15
|
+
/**
|
|
16
|
+
* The Judgment service (ADR 0017): evaluate typed questions against one
|
|
17
|
+
* State and return typed answers with probabilities and their origin. It
|
|
18
|
+
* answers; it never decides. Policy (thresholds, escalation, caching) lives
|
|
19
|
+
* in the flow package.
|
|
20
|
+
*/
|
|
21
|
+
|
|
22
|
+
export class JudgmentBackendError extends Schema.TaggedError<JudgmentBackendError>()(
|
|
23
|
+
"JudgmentBackendError",
|
|
24
|
+
{
|
|
25
|
+
backend: Schema.String,
|
|
26
|
+
message: Schema.String,
|
|
27
|
+
cause: Schema.optionalKey(Schema.Defect())
|
|
28
|
+
}
|
|
29
|
+
) {}
|
|
30
|
+
|
|
31
|
+
/** Raised by the typed accessors when a key is missing or of another kind. */
|
|
32
|
+
export class AnswerMismatch extends Schema.TaggedError<AnswerMismatch>()("AnswerMismatch", {
|
|
33
|
+
key: Schema.String,
|
|
34
|
+
expected: Schema.String,
|
|
35
|
+
actual: Schema.optionalKey(Schema.String),
|
|
36
|
+
reason: Schema.optionalKey(Schema.String)
|
|
37
|
+
}) {
|
|
38
|
+
get message(): string {
|
|
39
|
+
return this.actual === undefined
|
|
40
|
+
? `no ${this.expected} answer for "${this.key}"${this.reason === undefined ? "" : `: ${this.reason}`}`
|
|
41
|
+
: `answer "${this.key}" is a ${this.actual}, not a ${this.expected}`
|
|
42
|
+
}
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
export const JudgmentError = Schema.Union([JudgmentBackendError, AnswerMismatch])
|
|
46
|
+
export type JudgmentError = typeof JudgmentError.Type
|
|
47
|
+
|
|
48
|
+
export interface JudgmentInput {
|
|
49
|
+
readonly state: State
|
|
50
|
+
readonly questions: Readonly<Record<string, Question>>
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
export interface JudgmentShape {
|
|
54
|
+
readonly backend: JudgmentBackend
|
|
55
|
+
/**
|
|
56
|
+
* Backend plus checkpoint, e.g. `llm:mlx-lm:/models/qwen3-4b` or
|
|
57
|
+
* `typesafe:jev-1.13.0`: what a cached answer is keyed on, so swapping a
|
|
58
|
+
* local model never reuses its predecessor's answers.
|
|
59
|
+
*/
|
|
60
|
+
readonly identity: string
|
|
61
|
+
/**
|
|
62
|
+
* Answer every question independently over the same state. A question the
|
|
63
|
+
* backend could not answer lands in `failures`; the request as a whole
|
|
64
|
+
* fails only when the backend itself is unreachable or rejects it.
|
|
65
|
+
*/
|
|
66
|
+
readonly judge: (input: JudgmentInput) => Effect.Effect<JudgmentResult, JudgmentBackendError>
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
export class Judgment extends Context.Service<Judgment, JudgmentShape>()(
|
|
70
|
+
"@llm4ts/core/judgment/Judgment"
|
|
71
|
+
) {}
|
|
72
|
+
|
|
73
|
+
const answerOf = (
|
|
74
|
+
result: JudgmentResult,
|
|
75
|
+
key: string,
|
|
76
|
+
expected: Answer["type"]
|
|
77
|
+
): Effect.Effect<Answer, AnswerMismatch> => {
|
|
78
|
+
const answer = result.answers[key]
|
|
79
|
+
if (answer === undefined) {
|
|
80
|
+
const failure = result.failures.find((entry) => entry.key === key)
|
|
81
|
+
return Effect.fail(
|
|
82
|
+
AnswerMismatch.make({
|
|
83
|
+
key,
|
|
84
|
+
expected,
|
|
85
|
+
...(failure === undefined ? {} : { reason: failure.reason })
|
|
86
|
+
})
|
|
87
|
+
)
|
|
88
|
+
}
|
|
89
|
+
return answer.type === expected
|
|
90
|
+
? Effect.succeed(answer)
|
|
91
|
+
: Effect.fail(AnswerMismatch.make({ key, expected, actual: answer.type }))
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
/** Typed accessors: the honest way to read an answer by key. */
|
|
95
|
+
export const choiceOf = (
|
|
96
|
+
result: JudgmentResult,
|
|
97
|
+
key: string
|
|
98
|
+
): Effect.Effect<ChoiceAnswer, AnswerMismatch> =>
|
|
99
|
+
Effect.flatMap(answerOf(result, key, "choice"), (answer) =>
|
|
100
|
+
answer instanceof ChoiceAnswer
|
|
101
|
+
? Effect.succeed(answer)
|
|
102
|
+
: Effect.fail(AnswerMismatch.make({ key, expected: "choice", actual: answer.type }))
|
|
103
|
+
)
|
|
104
|
+
|
|
105
|
+
export const scoreOf = (
|
|
106
|
+
result: JudgmentResult,
|
|
107
|
+
key: string
|
|
108
|
+
): Effect.Effect<ScoreAnswer, AnswerMismatch> =>
|
|
109
|
+
Effect.flatMap(answerOf(result, key, "score"), (answer) =>
|
|
110
|
+
answer instanceof ScoreAnswer
|
|
111
|
+
? Effect.succeed(answer)
|
|
112
|
+
: Effect.fail(AnswerMismatch.make({ key, expected: "score", actual: answer.type }))
|
|
113
|
+
)
|
|
114
|
+
|
|
115
|
+
export const truthOf = (
|
|
116
|
+
result: JudgmentResult,
|
|
117
|
+
key: string
|
|
118
|
+
): Effect.Effect<TruthAnswer, AnswerMismatch> =>
|
|
119
|
+
Effect.flatMap(answerOf(result, key, "truth"), (answer) =>
|
|
120
|
+
answer instanceof TruthAnswer
|
|
121
|
+
? Effect.succeed(answer)
|
|
122
|
+
: Effect.fail(AnswerMismatch.make({ key, expected: "truth", actual: answer.type }))
|
|
123
|
+
)
|