@llm4ts/core 2.5.0 → 2.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/Connector.d.ts +4 -0
- package/dist/Connector.d.ts.map +1 -1
- package/dist/Connector.js +8 -3
- package/dist/Connector.js.map +1 -1
- package/dist/ConnectorConfig.d.ts.map +1 -1
- package/dist/ConnectorConfig.js +2 -0
- package/dist/ConnectorConfig.js.map +1 -1
- package/dist/ContextManagement.d.ts.map +1 -1
- package/dist/ContextManagement.js +1 -0
- package/dist/ContextManagement.js.map +1 -1
- package/dist/LabelScoring.d.ts +52 -0
- package/dist/LabelScoring.d.ts.map +1 -0
- package/dist/LabelScoring.js +139 -0
- package/dist/LabelScoring.js.map +1 -0
- package/dist/LlmService.d.ts +30 -2
- package/dist/LlmService.d.ts.map +1 -1
- package/dist/LlmService.js.map +1 -1
- package/dist/Models.d.ts +34 -2
- package/dist/Models.d.ts.map +1 -1
- package/dist/Models.js +37 -1
- package/dist/Models.js.map +1 -1
- package/dist/eval/Judge.d.ts +15 -0
- package/dist/eval/Judge.d.ts.map +1 -1
- package/dist/eval/Judge.js +50 -0
- package/dist/eval/Judge.js.map +1 -1
- package/dist/judgment/FakeJudgment.d.ts +22 -0
- package/dist/judgment/FakeJudgment.d.ts.map +1 -0
- package/dist/judgment/FakeJudgment.js +42 -0
- package/dist/judgment/FakeJudgment.js.map +1 -0
- package/dist/judgment/Judgment.d.ts +57 -0
- package/dist/judgment/Judgment.d.ts.map +1 -0
- package/dist/judgment/Judgment.js +57 -0
- package/dist/judgment/Judgment.js.map +1 -0
- package/dist/judgment/LlmJudgment.d.ts +73 -0
- package/dist/judgment/LlmJudgment.d.ts.map +1 -0
- package/dist/judgment/LlmJudgment.js +239 -0
- package/dist/judgment/LlmJudgment.js.map +1 -0
- package/dist/judgment/Schemas.d.ts +163 -0
- package/dist/judgment/Schemas.d.ts.map +1 -0
- package/dist/judgment/Schemas.js +197 -0
- package/dist/judgment/Schemas.js.map +1 -0
- package/dist/judgment/TypeSafeJudgment.d.ts +46 -0
- package/dist/judgment/TypeSafeJudgment.d.ts.map +1 -0
- package/dist/judgment/TypeSafeJudgment.js +185 -0
- package/dist/judgment/TypeSafeJudgment.js.map +1 -0
- package/dist/observability/MeteredLlmService.d.ts.map +1 -1
- package/dist/observability/MeteredLlmService.js +1 -0
- package/dist/observability/MeteredLlmService.js.map +1 -1
- package/dist/providers/ConnectorFactories.d.ts.map +1 -1
- package/dist/providers/ConnectorFactories.js +2 -0
- package/dist/providers/ConnectorFactories.js.map +1 -1
- package/dist/providers/LmStudioProvider.d.ts.map +1 -1
- package/dist/providers/LmStudioProvider.js +45 -34
- package/dist/providers/LmStudioProvider.js.map +1 -1
- package/dist/providers/MlxLmProvider.d.ts +29 -0
- package/dist/providers/MlxLmProvider.d.ts.map +1 -0
- package/dist/providers/MlxLmProvider.js +249 -0
- package/dist/providers/MlxLmProvider.js.map +1 -0
- package/dist/providers/MockProvider.d.ts.map +1 -1
- package/dist/providers/MockProvider.js +11 -0
- package/dist/providers/MockProvider.js.map +1 -1
- package/dist/providers/OpenAIModels.d.ts +35 -0
- package/dist/providers/OpenAIModels.d.ts.map +1 -1
- package/dist/providers/OpenAIModels.js +36 -3
- package/dist/providers/OpenAIModels.js.map +1 -1
- package/package.json +8 -1
- package/src/Connector.ts +13 -3
- package/src/ConnectorConfig.ts +2 -0
- package/src/ContextManagement.ts +1 -0
- package/src/LabelScoring.ts +211 -0
- package/src/LlmService.ts +37 -1
- package/src/Models.ts +43 -1
- package/src/eval/Judge.ts +74 -0
- package/src/judgment/FakeJudgment.ts +90 -0
- package/src/judgment/Judgment.ts +123 -0
- package/src/judgment/LlmJudgment.ts +399 -0
- package/src/judgment/Schemas.ts +278 -0
- package/src/judgment/TypeSafeJudgment.ts +253 -0
- package/src/observability/MeteredLlmService.ts +7 -0
- package/src/providers/ConnectorFactories.ts +4 -0
- package/src/providers/LmStudioProvider.ts +64 -48
- package/src/providers/MlxLmProvider.ts +357 -0
- package/src/providers/MockProvider.ts +19 -0
- package/src/providers/OpenAIModels.ts +40 -3
|
@@ -0,0 +1,399 @@
|
|
|
1
|
+
import * as Effect from "effect/Effect"
|
|
2
|
+
import * as Layer from "effect/Layer"
|
|
3
|
+
import * as Result from "effect/Result"
|
|
4
|
+
import * as Schema from "effect/Schema"
|
|
5
|
+
import type { LlmError } from "../Errors.ts"
|
|
6
|
+
import { verbalizedScoreLabelSequence, verbalizedScoreLabels } from "../LabelScoring.ts"
|
|
7
|
+
import type { LabelSequence, LlmServiceShape } from "../LlmService.ts"
|
|
8
|
+
import { TokenUsage, type LabelDistribution } from "../Models.ts"
|
|
9
|
+
import { Judgment, type JudgmentInput, type JudgmentShape } from "./Judgment.ts"
|
|
10
|
+
import {
|
|
11
|
+
averageProbabilities,
|
|
12
|
+
choiceAnswer,
|
|
13
|
+
JudgmentResult,
|
|
14
|
+
QuestionFailure,
|
|
15
|
+
renderState,
|
|
16
|
+
scoreAnswer,
|
|
17
|
+
truthAnswer,
|
|
18
|
+
origins,
|
|
19
|
+
type Answer,
|
|
20
|
+
type AnswerOrigin,
|
|
21
|
+
type Description,
|
|
22
|
+
type Question,
|
|
23
|
+
type ScoringMethod,
|
|
24
|
+
type State
|
|
25
|
+
} from "./Schemas.ts"
|
|
26
|
+
|
|
27
|
+
/**
|
|
28
|
+
* `LlmJudgment`: the Judgment service over any `LlmServiceShape`. One label
|
|
29
|
+
* question per atomic question, answered through `scoreLabels`: a single
|
|
30
|
+
* forward pass on connectors with log-probabilities, a short JSON reply on
|
|
31
|
+
* the rest. Questions never see each other's answers in `independent`
|
|
32
|
+
* batching; `shared-prefix` sends the state once with every question and
|
|
33
|
+
* reads each answer as its own label distribution, trading that isolation
|
|
34
|
+
* for one prompt prefix per request.
|
|
35
|
+
*/
|
|
36
|
+
|
|
37
|
+
export const Batching = Schema.Literals(["independent", "shared-prefix"])
|
|
38
|
+
export type Batching = typeof Batching.Type
|
|
39
|
+
|
|
40
|
+
export class LlmJudgmentConfig extends Schema.Class<LlmJudgmentConfig>("LlmJudgmentConfig")({
|
|
41
|
+
/** Questions in flight at once. 1 keeps a local server's prompt cache warm. */
|
|
42
|
+
concurrency: Schema.Int.pipe(Schema.withConstructorDefault(Effect.succeed(1))),
|
|
43
|
+
/** 2 also asks with the options reversed and averages, to blunt position bias. */
|
|
44
|
+
permutations: Schema.Literals([1, 2]).pipe(
|
|
45
|
+
Schema.withConstructorDefault(Effect.succeed<1 | 2>(1))
|
|
46
|
+
),
|
|
47
|
+
/**
|
|
48
|
+
* `independent` (default): one call per question, Jev's answer
|
|
49
|
+
* independence. `shared-prefix`: every question of a request in one call
|
|
50
|
+
* whose state prefix is sent once; a question the batch could not read
|
|
51
|
+
* falls back to its own independent call.
|
|
52
|
+
*/
|
|
53
|
+
batching: Batching.pipe(Schema.withConstructorDefault(Effect.succeed<Batching>("independent"))),
|
|
54
|
+
/** Connector and model behind the seat, for `identity` and answer origins. */
|
|
55
|
+
connector: Schema.optionalKey(Schema.String),
|
|
56
|
+
model: Schema.optionalKey(Schema.String)
|
|
57
|
+
}) {}
|
|
58
|
+
|
|
59
|
+
export interface LlmJudgmentHooks {
|
|
60
|
+
/** Called once per request with the summed backend usage, when any was reported. */
|
|
61
|
+
readonly onUsage?: (usage: TokenUsage, model: string | undefined) => Effect.Effect<void>
|
|
62
|
+
/** Called when a question fell back from label scoring to the verbalized path. */
|
|
63
|
+
readonly onFallback?: (key: string, reason: string) => Effect.Effect<void>
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
const LABELS = "ABCDEFGHIJKLMNOPQRSTUVWXYZ"
|
|
67
|
+
|
|
68
|
+
const describe = (description: Description): string =>
|
|
69
|
+
typeof description === "string" ? description : JSON.stringify(description)
|
|
70
|
+
|
|
71
|
+
interface LabelPlan {
|
|
72
|
+
readonly prompt: string
|
|
73
|
+
readonly labels: ReadonlyArray<string>
|
|
74
|
+
/** Label to the option key or level index it stands for. */
|
|
75
|
+
readonly keys: Readonly<Record<string, string>>
|
|
76
|
+
}
|
|
77
|
+
|
|
78
|
+
/** A question's part of a prompt, without the state header. */
|
|
79
|
+
export interface QuestionBody {
|
|
80
|
+
readonly body: string
|
|
81
|
+
readonly labels: ReadonlyArray<string>
|
|
82
|
+
readonly keys: Readonly<Record<string, string>>
|
|
83
|
+
}
|
|
84
|
+
|
|
85
|
+
export const questionBody = (question: Question, reverse: boolean): QuestionBody => {
|
|
86
|
+
if (question.type === "truth") {
|
|
87
|
+
const criteria =
|
|
88
|
+
question.criteria === undefined
|
|
89
|
+
? ""
|
|
90
|
+
: `\nYes means: ${describe(question.criteria.true)}\nNo means: ${describe(question.criteria.false)}`
|
|
91
|
+
return {
|
|
92
|
+
body: `Statement: ${question.instructions}${criteria}\nAnswer with exactly one word: yes or no.`,
|
|
93
|
+
labels: ["yes", "no"],
|
|
94
|
+
keys: { yes: "yes", no: "no" }
|
|
95
|
+
}
|
|
96
|
+
}
|
|
97
|
+
const entries: ReadonlyArray<readonly [key: string, description: Description]> =
|
|
98
|
+
question.type === "choice"
|
|
99
|
+
? Object.entries(question.criteria)
|
|
100
|
+
: question.criteria.map((level, index) => [String(index), level] as const)
|
|
101
|
+
const ordered = reverse ? [...entries].reverse() : entries
|
|
102
|
+
const labels = ordered.map((_, index) => LABELS[index] ?? `L${index}`)
|
|
103
|
+
const lines = ordered
|
|
104
|
+
.map(([key, description], index) =>
|
|
105
|
+
question.type === "choice"
|
|
106
|
+
? `${labels[index]}. ${key}: ${describe(description)}`
|
|
107
|
+
: `${labels[index]}. ${describe(description)}`
|
|
108
|
+
)
|
|
109
|
+
.join("\n")
|
|
110
|
+
const ask = question.type === "choice" ? "Question" : "Rate on the levels below"
|
|
111
|
+
return {
|
|
112
|
+
body: `${ask}: ${question.instructions}\nOptions:\n${lines}\nAnswer with exactly one letter.`,
|
|
113
|
+
labels,
|
|
114
|
+
keys: Object.fromEntries(ordered.map(([key], index) => [labels[index] ?? `L${index}`, key]))
|
|
115
|
+
}
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
const stateHeader = (state: State): string => `State:\n${renderState(state)}\n\n`
|
|
119
|
+
|
|
120
|
+
/**
|
|
121
|
+
* The fixed label-scoring template: state first, then the question, then
|
|
122
|
+
* one single-token label per option. Truth uses `yes`/`no`.
|
|
123
|
+
*/
|
|
124
|
+
export const labelPlan = (state: State, question: Question, reverse: boolean): LabelPlan => {
|
|
125
|
+
const part = questionBody(question, reverse)
|
|
126
|
+
return { prompt: `${stateHeader(state)}${part.body}`, labels: part.labels, keys: part.keys }
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
/**
|
|
130
|
+
* The shared-prefix template: the state once, then every question numbered,
|
|
131
|
+
* and an answer format of one label per line so a backend can read each
|
|
132
|
+
* position as its own distribution.
|
|
133
|
+
*/
|
|
134
|
+
export const sequencePlan = (
|
|
135
|
+
state: State,
|
|
136
|
+
questions: ReadonlyArray<Question>,
|
|
137
|
+
reverse: boolean
|
|
138
|
+
): { readonly prompt: string; readonly parts: ReadonlyArray<QuestionBody> } => {
|
|
139
|
+
const parts = questions.map((question) => questionBody(question, reverse))
|
|
140
|
+
const numbered = parts.map((part, index) => `Question ${index + 1}.\n${part.body}`).join("\n\n")
|
|
141
|
+
return {
|
|
142
|
+
prompt:
|
|
143
|
+
`${stateHeader(state)}Answer each numbered question below with exactly one label on its own line, ` +
|
|
144
|
+
`formatted as "<number>: <label>", in order, and nothing else.\n\n${numbered}`,
|
|
145
|
+
parts
|
|
146
|
+
}
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
const byKey = (
|
|
150
|
+
keys: Readonly<Record<string, string>>,
|
|
151
|
+
distribution: LabelDistribution
|
|
152
|
+
): Record<string, number> =>
|
|
153
|
+
Object.fromEntries(
|
|
154
|
+
Object.entries(distribution.probabilities).map(([label, value]) => [
|
|
155
|
+
keys[label] ?? label,
|
|
156
|
+
value
|
|
157
|
+
])
|
|
158
|
+
)
|
|
159
|
+
|
|
160
|
+
const toAnswer = (
|
|
161
|
+
question: Question,
|
|
162
|
+
probabilities: Readonly<Record<string, number>>,
|
|
163
|
+
origin: AnswerOrigin,
|
|
164
|
+
support: number
|
|
165
|
+
): Answer =>
|
|
166
|
+
question.type === "choice"
|
|
167
|
+
? choiceAnswer(probabilities, origin, support)
|
|
168
|
+
: question.type === "score"
|
|
169
|
+
? scoreAnswer(question, probabilities, origin, support)
|
|
170
|
+
: truthAnswer(probabilities["yes"] ?? 0, origin, support)
|
|
171
|
+
|
|
172
|
+
const sumUsage = (usages: ReadonlyArray<TokenUsage | undefined>): TokenUsage | undefined => {
|
|
173
|
+
const present = usages.filter((usage): usage is TokenUsage => usage !== undefined)
|
|
174
|
+
return present.length === 0
|
|
175
|
+
? undefined
|
|
176
|
+
: TokenUsage.make({
|
|
177
|
+
prompt: present.reduce((sum, usage) => sum + usage.prompt, 0),
|
|
178
|
+
completion: present.reduce((sum, usage) => sum + usage.completion, 0),
|
|
179
|
+
total: present.reduce((sum, usage) => sum + usage.total, 0)
|
|
180
|
+
})
|
|
181
|
+
}
|
|
182
|
+
|
|
183
|
+
interface Scored {
|
|
184
|
+
readonly answer: Answer
|
|
185
|
+
readonly usage: TokenUsage | undefined
|
|
186
|
+
readonly model: string | undefined
|
|
187
|
+
}
|
|
188
|
+
|
|
189
|
+
type Read = readonly [keys: Readonly<Record<string, string>>, distribution: LabelDistribution]
|
|
190
|
+
|
|
191
|
+
/** Combine one distribution per permutation into an answer; the weaker method and support win. */
|
|
192
|
+
const combine = (
|
|
193
|
+
question: Question,
|
|
194
|
+
reads: ReadonlyArray<Read>,
|
|
195
|
+
fallbackModel: string | undefined
|
|
196
|
+
): Scored => {
|
|
197
|
+
const method: ScoringMethod = reads.every(
|
|
198
|
+
([, distribution]) => distribution.method === "logprobs"
|
|
199
|
+
)
|
|
200
|
+
? "logprobs"
|
|
201
|
+
: reads.some(([, distribution]) => distribution.method === "sampled")
|
|
202
|
+
? "sampled"
|
|
203
|
+
: "verbalized"
|
|
204
|
+
const support = Math.min(...reads.map(([, distribution]) => distribution.support))
|
|
205
|
+
const probabilities = averageProbabilities(
|
|
206
|
+
reads.map(([keys, distribution]) => byKey(keys, distribution))
|
|
207
|
+
)
|
|
208
|
+
const model =
|
|
209
|
+
reads.find(([, distribution]) => distribution.model !== undefined)?.[1].model ?? fallbackModel
|
|
210
|
+
return {
|
|
211
|
+
answer: toAnswer(question, probabilities, origins.llm(method, model), support),
|
|
212
|
+
usage: sumUsage(reads.map(([, distribution]) => distribution.usage)),
|
|
213
|
+
model
|
|
214
|
+
}
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
type Outcome = readonly [key: string, outcome: Result.Result<Scored, LlmError>]
|
|
218
|
+
|
|
219
|
+
export const makeLlmJudgment = (
|
|
220
|
+
llm: LlmServiceShape,
|
|
221
|
+
config: LlmJudgmentConfig = LlmJudgmentConfig.make({}),
|
|
222
|
+
hooks: LlmJudgmentHooks = {}
|
|
223
|
+
): JudgmentShape => {
|
|
224
|
+
const fallback = verbalizedScoreLabels(llm.executeStructuredWithUsage)
|
|
225
|
+
const sequence =
|
|
226
|
+
llm.scoreLabelSequence ?? verbalizedScoreLabelSequence(llm.executeStructuredWithUsage)
|
|
227
|
+
const permutations: ReadonlyArray<boolean> = config.permutations === 2 ? [false, true] : [false]
|
|
228
|
+
const permutationsFor = (question: Question): ReadonlyArray<boolean> =>
|
|
229
|
+
permutations.filter((reverse) => !reverse || question.type !== "truth")
|
|
230
|
+
|
|
231
|
+
const scoreOnce = (plan: LabelPlan, key: string): Effect.Effect<LabelDistribution, LlmError> =>
|
|
232
|
+
llm
|
|
233
|
+
.scoreLabels(plan.prompt, plan.labels)
|
|
234
|
+
.pipe(
|
|
235
|
+
Effect.catchTag("ParseError", (error) =>
|
|
236
|
+
(hooks.onFallback?.(key, error.message) ?? Effect.void).pipe(
|
|
237
|
+
Effect.andThen(fallback(plan.prompt, plan.labels))
|
|
238
|
+
)
|
|
239
|
+
)
|
|
240
|
+
)
|
|
241
|
+
|
|
242
|
+
const scoreQuestion = (
|
|
243
|
+
state: State,
|
|
244
|
+
key: string,
|
|
245
|
+
question: Question
|
|
246
|
+
): Effect.Effect<Scored, LlmError> =>
|
|
247
|
+
Effect.gen(function* () {
|
|
248
|
+
const plans = permutationsFor(question).map((reverse) => labelPlan(state, question, reverse))
|
|
249
|
+
const reads = yield* Effect.forEach(plans, (plan) =>
|
|
250
|
+
Effect.map(scoreOnce(plan, key), (distribution): Read => [plan.keys, distribution])
|
|
251
|
+
)
|
|
252
|
+
return combine(question, reads, config.model)
|
|
253
|
+
})
|
|
254
|
+
|
|
255
|
+
const independently = (
|
|
256
|
+
input: JudgmentInput,
|
|
257
|
+
entries: ReadonlyArray<readonly [string, Question]>
|
|
258
|
+
): Effect.Effect<ReadonlyArray<Outcome>> =>
|
|
259
|
+
Effect.forEach(
|
|
260
|
+
entries,
|
|
261
|
+
([key, question]) =>
|
|
262
|
+
Effect.map(
|
|
263
|
+
Effect.result(scoreQuestion(input.state, key, question)),
|
|
264
|
+
(outcome): Outcome => [key, outcome]
|
|
265
|
+
),
|
|
266
|
+
{ concurrency: config.concurrency }
|
|
267
|
+
)
|
|
268
|
+
|
|
269
|
+
/**
|
|
270
|
+
* One call per permutation for the whole request. A question whose
|
|
271
|
+
* position could not be read in any permutation, or a call that failed
|
|
272
|
+
* outright, goes back through the independent path for that question.
|
|
273
|
+
*/
|
|
274
|
+
const sharedPrefix = (
|
|
275
|
+
input: JudgmentInput,
|
|
276
|
+
entries: ReadonlyArray<readonly [string, Question]>
|
|
277
|
+
): Effect.Effect<{
|
|
278
|
+
readonly outcomes: ReadonlyArray<Outcome>
|
|
279
|
+
readonly usage: TokenUsage | undefined
|
|
280
|
+
readonly model: string | undefined
|
|
281
|
+
}> =>
|
|
282
|
+
Effect.gen(function* () {
|
|
283
|
+
const questions = entries.map(([, question]) => question)
|
|
284
|
+
const calls = yield* Effect.forEach(permutations, (reverse) => {
|
|
285
|
+
const plan = sequencePlan(input.state, questions, reverse)
|
|
286
|
+
return Effect.map(
|
|
287
|
+
Effect.result(
|
|
288
|
+
sequence(
|
|
289
|
+
plan.prompt,
|
|
290
|
+
plan.parts.map((part) => part.labels)
|
|
291
|
+
)
|
|
292
|
+
),
|
|
293
|
+
(outcome) => [plan, outcome] as const
|
|
294
|
+
)
|
|
295
|
+
})
|
|
296
|
+
const successes = calls.flatMap(([, outcome]) =>
|
|
297
|
+
Result.isSuccess(outcome) ? [outcome.success] : []
|
|
298
|
+
)
|
|
299
|
+
const batchModel = successes.find((call) => call.model !== undefined)?.model ?? config.model
|
|
300
|
+
const answered: Array<Outcome> = []
|
|
301
|
+
const retry: Array<readonly [string, Question]> = []
|
|
302
|
+
for (const [index, [key, question]] of entries.entries()) {
|
|
303
|
+
const reads: Array<Read> = []
|
|
304
|
+
let reason: string | undefined
|
|
305
|
+
for (const [callIndex, reverse] of permutations.entries()) {
|
|
306
|
+
if (!permutationsFor(question).includes(reverse)) {
|
|
307
|
+
continue
|
|
308
|
+
}
|
|
309
|
+
const call = calls[callIndex]
|
|
310
|
+
const part = call?.[0].parts[index]
|
|
311
|
+
const outcome = call?.[1]
|
|
312
|
+
if (outcome === undefined || part === undefined) {
|
|
313
|
+
reason = "no batched call covered this question"
|
|
314
|
+
break
|
|
315
|
+
}
|
|
316
|
+
if (Result.isFailure(outcome)) {
|
|
317
|
+
reason = outcome.failure.message
|
|
318
|
+
break
|
|
319
|
+
}
|
|
320
|
+
const entry = outcome.success.entries[index]
|
|
321
|
+
if (entry === undefined) {
|
|
322
|
+
reason = "no entry for this question"
|
|
323
|
+
break
|
|
324
|
+
}
|
|
325
|
+
if (Result.isFailure(entry)) {
|
|
326
|
+
reason = entry.failure.message
|
|
327
|
+
break
|
|
328
|
+
}
|
|
329
|
+
reads.push([part.keys, entry.success])
|
|
330
|
+
}
|
|
331
|
+
if (reason === undefined && reads.length > 0) {
|
|
332
|
+
// Usage was reported once for the whole call; the per-question reads carry none.
|
|
333
|
+
answered.push([key, Result.succeed(combine(question, reads, batchModel))])
|
|
334
|
+
} else {
|
|
335
|
+
yield* hooks.onFallback?.(key, `shared-prefix batch: ${reason ?? "unreadable"}`) ??
|
|
336
|
+
Effect.void
|
|
337
|
+
retry.push([key, question])
|
|
338
|
+
}
|
|
339
|
+
}
|
|
340
|
+
const retried = retry.length === 0 ? [] : yield* independently(input, retry)
|
|
341
|
+
const order = new Map(entries.map(([key], index) => [key, index] as const))
|
|
342
|
+
const outcomes = [...answered, ...retried].sort(
|
|
343
|
+
([a], [b]) => (order.get(a) ?? 0) - (order.get(b) ?? 0)
|
|
344
|
+
)
|
|
345
|
+
return {
|
|
346
|
+
outcomes,
|
|
347
|
+
usage: sumUsage(successes.map((call) => call.usage)),
|
|
348
|
+
model: successes.find((call) => call.model !== undefined)?.model
|
|
349
|
+
}
|
|
350
|
+
})
|
|
351
|
+
|
|
352
|
+
const judge = Effect.fn("@llm4ts/core/judgment/LlmJudgment.judge")(function* (
|
|
353
|
+
input: JudgmentInput
|
|
354
|
+
): Effect.fn.Return<JudgmentResult> {
|
|
355
|
+
const entries = Object.entries(input.questions)
|
|
356
|
+
const batched =
|
|
357
|
+
config.batching === "shared-prefix" && entries.length > 1
|
|
358
|
+
? yield* sharedPrefix(input, entries)
|
|
359
|
+
: { outcomes: yield* independently(input, entries), usage: undefined, model: undefined }
|
|
360
|
+
const answers: Record<string, Answer> = {}
|
|
361
|
+
const failures: Array<QuestionFailure> = []
|
|
362
|
+
const usages: Array<TokenUsage | undefined> = [batched.usage]
|
|
363
|
+
let model: string | undefined = batched.model
|
|
364
|
+
for (const [key, outcome] of batched.outcomes) {
|
|
365
|
+
if (Result.isFailure(outcome)) {
|
|
366
|
+
failures.push(QuestionFailure.make({ key, reason: outcome.failure.message }))
|
|
367
|
+
} else {
|
|
368
|
+
answers[key] = outcome.success.answer
|
|
369
|
+
usages.push(outcome.success.usage)
|
|
370
|
+
model = model ?? outcome.success.model
|
|
371
|
+
}
|
|
372
|
+
}
|
|
373
|
+
const usage = sumUsage(usages)
|
|
374
|
+
if (usage !== undefined && hooks.onUsage !== undefined) {
|
|
375
|
+
yield* hooks.onUsage(usage, model)
|
|
376
|
+
}
|
|
377
|
+
return JudgmentResult.make({
|
|
378
|
+
answers,
|
|
379
|
+
failures,
|
|
380
|
+
backend: "llm",
|
|
381
|
+
...(usage === undefined ? {} : { usage }),
|
|
382
|
+
...(model === undefined ? {} : { model })
|
|
383
|
+
})
|
|
384
|
+
})
|
|
385
|
+
|
|
386
|
+
return {
|
|
387
|
+
backend: "llm",
|
|
388
|
+
identity: `llm:${config.connector ?? "unknown"}:${config.model ?? "default"}`,
|
|
389
|
+
judge
|
|
390
|
+
}
|
|
391
|
+
}
|
|
392
|
+
|
|
393
|
+
export const LlmJudgmentLive = (
|
|
394
|
+
llm: LlmServiceShape,
|
|
395
|
+
config?: LlmJudgmentConfig,
|
|
396
|
+
hooks?: LlmJudgmentHooks
|
|
397
|
+
): Layer.Layer<Judgment> => Layer.succeed(Judgment, makeLlmJudgment(llm, config, hooks))
|
|
398
|
+
|
|
399
|
+
export type { LabelSequence }
|
|
@@ -0,0 +1,278 @@
|
|
|
1
|
+
import * as Effect from "effect/Effect"
|
|
2
|
+
import * as Schema from "effect/Schema"
|
|
3
|
+
import { TokenUsage } from "../Models.ts"
|
|
4
|
+
|
|
5
|
+
/**
|
|
6
|
+
* Typed judgments (ADR 0017): atomic questions evaluated against one State,
|
|
7
|
+
* answered with probabilities instead of text. Wire names mirror TypeSafe's
|
|
8
|
+
* Jev so its cookbooks port by search and replace, with two changes: `noul`
|
|
9
|
+
* is `truth`, and every answer carries an `origin` (backend, checkpoint,
|
|
10
|
+
* extraction method, calibration evidence, escalation) plus `support`, the
|
|
11
|
+
* mass the backend actually placed on the offered options. Vocabulary:
|
|
12
|
+
* CONTEXT.md.
|
|
13
|
+
*/
|
|
14
|
+
|
|
15
|
+
/** The content a judgment evaluates. Data, never instructions. */
|
|
16
|
+
export const State = Schema.Union([
|
|
17
|
+
Schema.String,
|
|
18
|
+
Schema.Record(Schema.String, Schema.Json),
|
|
19
|
+
Schema.Array(Schema.String)
|
|
20
|
+
])
|
|
21
|
+
export type State = typeof State.Type
|
|
22
|
+
|
|
23
|
+
/** An option, level, or criterion: text or any JSON structure (Jev's `EntryType`). */
|
|
24
|
+
export const Description = Schema.Json
|
|
25
|
+
export type Description = typeof Description.Type
|
|
26
|
+
|
|
27
|
+
export class ChoiceQuestion extends Schema.Class<ChoiceQuestion>("ChoiceQuestion")({
|
|
28
|
+
type: Schema.Literal("choice"),
|
|
29
|
+
instructions: Schema.String,
|
|
30
|
+
/** Option key to description. Include an "other" key when coverage is uncertain. */
|
|
31
|
+
criteria: Schema.Record(Schema.String, Description)
|
|
32
|
+
}) {}
|
|
33
|
+
|
|
34
|
+
export class ScoreQuestion extends Schema.Class<ScoreQuestion>("ScoreQuestion")({
|
|
35
|
+
type: Schema.Literal("score"),
|
|
36
|
+
instructions: Schema.String,
|
|
37
|
+
/** Ordered levels, index 0 first. */
|
|
38
|
+
criteria: Schema.Array(Description)
|
|
39
|
+
}) {}
|
|
40
|
+
|
|
41
|
+
export class TruthCriteria extends Schema.Class<TruthCriteria>("TruthCriteria")({
|
|
42
|
+
true: Description,
|
|
43
|
+
false: Description
|
|
44
|
+
}) {}
|
|
45
|
+
|
|
46
|
+
export class TruthQuestion extends Schema.Class<TruthQuestion>("TruthQuestion")({
|
|
47
|
+
type: Schema.Literal("truth"),
|
|
48
|
+
instructions: Schema.String,
|
|
49
|
+
criteria: Schema.optionalKey(TruthCriteria)
|
|
50
|
+
}) {}
|
|
51
|
+
|
|
52
|
+
export const Question = Schema.Union([ChoiceQuestion, ScoreQuestion, TruthQuestion])
|
|
53
|
+
export type Question = typeof Question.Type
|
|
54
|
+
|
|
55
|
+
export const choice = (
|
|
56
|
+
instructions: string,
|
|
57
|
+
criteria: Readonly<Record<string, Description>>
|
|
58
|
+
): ChoiceQuestion => ChoiceQuestion.make({ type: "choice", instructions, criteria })
|
|
59
|
+
|
|
60
|
+
export const score = (instructions: string, criteria: ReadonlyArray<Description>): ScoreQuestion =>
|
|
61
|
+
ScoreQuestion.make({ type: "score", instructions, criteria })
|
|
62
|
+
|
|
63
|
+
export const truth = (
|
|
64
|
+
instructions: string,
|
|
65
|
+
criteria?: { readonly true: Description; readonly false: Description }
|
|
66
|
+
): TruthQuestion =>
|
|
67
|
+
TruthQuestion.make({
|
|
68
|
+
type: "truth",
|
|
69
|
+
instructions,
|
|
70
|
+
...(criteria === undefined ? {} : { criteria: TruthCriteria.make(criteria) })
|
|
71
|
+
})
|
|
72
|
+
|
|
73
|
+
export const JudgmentBackend = Schema.Literals(["typesafe", "llm", "fake"])
|
|
74
|
+
export type JudgmentBackend = typeof JudgmentBackend.Type
|
|
75
|
+
|
|
76
|
+
/** How the probabilities were extracted. */
|
|
77
|
+
export const ScoringMethod = Schema.Literals([
|
|
78
|
+
"logprobs",
|
|
79
|
+
"verbalized",
|
|
80
|
+
"sampled",
|
|
81
|
+
"reasoning",
|
|
82
|
+
"hosted"
|
|
83
|
+
])
|
|
84
|
+
export type ScoringMethod = typeof ScoringMethod.Type
|
|
85
|
+
|
|
86
|
+
/**
|
|
87
|
+
* What is known about the numbers' calibration: `none` (nothing), `claimed`
|
|
88
|
+
* (the provider says so; TypeSafe), `measured` (an evaluation in this
|
|
89
|
+
* project produced the evidence; see the judgment decision map).
|
|
90
|
+
*/
|
|
91
|
+
export const Calibration = Schema.Literals(["none", "claimed", "measured"])
|
|
92
|
+
export type Calibration = typeof Calibration.Type
|
|
93
|
+
|
|
94
|
+
/**
|
|
95
|
+
* Where an answer came from, in four separate facts a policy may key on:
|
|
96
|
+
* which backend and checkpoint produced it, how the probabilities were
|
|
97
|
+
* extracted, what is known about their calibration, and whether the answer
|
|
98
|
+
* replaced an earlier one through escalation.
|
|
99
|
+
*/
|
|
100
|
+
export class AnswerOrigin extends Schema.Class<AnswerOrigin>("AnswerOrigin")({
|
|
101
|
+
backend: JudgmentBackend,
|
|
102
|
+
model: Schema.optionalKey(Schema.String),
|
|
103
|
+
method: ScoringMethod,
|
|
104
|
+
calibration: Calibration.pipe(
|
|
105
|
+
Schema.withConstructorDefault(Effect.succeed<Calibration>("none")),
|
|
106
|
+
Schema.withDecodingDefaultKey(Effect.succeed<Calibration>("none"))
|
|
107
|
+
),
|
|
108
|
+
escalated: Schema.Boolean.pipe(
|
|
109
|
+
Schema.withConstructorDefault(Effect.succeed(false)),
|
|
110
|
+
Schema.withDecodingDefaultKey(Effect.succeed(false))
|
|
111
|
+
)
|
|
112
|
+
}) {}
|
|
113
|
+
|
|
114
|
+
/**
|
|
115
|
+
* Probability mass the backend placed on the offered options before
|
|
116
|
+
* renormalization (1 when declared over the options alone). Low support
|
|
117
|
+
* means the distribution was rebuilt from a sliver and must not be acted on.
|
|
118
|
+
*/
|
|
119
|
+
const Support = Schema.Number.pipe(
|
|
120
|
+
Schema.withConstructorDefault(Effect.succeed(1)),
|
|
121
|
+
Schema.withDecodingDefaultKey(Effect.succeed(1))
|
|
122
|
+
)
|
|
123
|
+
|
|
124
|
+
export class ChoiceAnswer extends Schema.Class<ChoiceAnswer>("ChoiceAnswer")({
|
|
125
|
+
type: Schema.Literal("choice"),
|
|
126
|
+
choice: Schema.String,
|
|
127
|
+
probabilities: Schema.Record(Schema.String, Schema.Number),
|
|
128
|
+
/** The maximum probability, everywhere: the same statistic whatever the backend. */
|
|
129
|
+
confidence: Schema.Number,
|
|
130
|
+
/** The backend's own confidence statistic when it reports one (TypeSafe), kept for comparison. */
|
|
131
|
+
reportedConfidence: Schema.optionalKey(Schema.Number),
|
|
132
|
+
support: Support,
|
|
133
|
+
origin: AnswerOrigin
|
|
134
|
+
}) {}
|
|
135
|
+
|
|
136
|
+
export class ScoreAnswer extends Schema.Class<ScoreAnswer>("ScoreAnswer")({
|
|
137
|
+
type: Schema.Literal("score"),
|
|
138
|
+
/** Expected level index; may fall between two levels. */
|
|
139
|
+
score: Schema.Number,
|
|
140
|
+
/** Level index (as a string key) to its description, echoed from the question. */
|
|
141
|
+
legend: Schema.Record(Schema.String, Description),
|
|
142
|
+
probabilities: Schema.Record(Schema.String, Schema.Number),
|
|
143
|
+
confidence: Schema.Number,
|
|
144
|
+
reportedConfidence: Schema.optionalKey(Schema.Number),
|
|
145
|
+
support: Support,
|
|
146
|
+
origin: AnswerOrigin
|
|
147
|
+
}) {}
|
|
148
|
+
|
|
149
|
+
export class TruthAnswer extends Schema.Class<TruthAnswer>("TruthAnswer")({
|
|
150
|
+
type: Schema.Literal("truth"),
|
|
151
|
+
/** Probability that the statement holds. */
|
|
152
|
+
truth: Schema.Number,
|
|
153
|
+
support: Support,
|
|
154
|
+
origin: AnswerOrigin
|
|
155
|
+
}) {}
|
|
156
|
+
|
|
157
|
+
export const Answer = Schema.Union([ChoiceAnswer, ScoreAnswer, TruthAnswer])
|
|
158
|
+
export type Answer = typeof Answer.Type
|
|
159
|
+
|
|
160
|
+
/** The answer type a question kind produces, for typed call sites. */
|
|
161
|
+
export type AnswerFor<Q extends Question> = Q extends ChoiceQuestion
|
|
162
|
+
? ChoiceAnswer
|
|
163
|
+
: Q extends ScoreQuestion
|
|
164
|
+
? ScoreAnswer
|
|
165
|
+
: TruthAnswer
|
|
166
|
+
|
|
167
|
+
/** A question that could not be answered; the others in the request still were. */
|
|
168
|
+
export class QuestionFailure extends Schema.Class<QuestionFailure>("QuestionFailure")({
|
|
169
|
+
key: Schema.String,
|
|
170
|
+
reason: Schema.String
|
|
171
|
+
}) {}
|
|
172
|
+
|
|
173
|
+
export class JudgmentRequest extends Schema.Class<JudgmentRequest>("JudgmentRequest")({
|
|
174
|
+
state: State,
|
|
175
|
+
questions: Schema.Record(Schema.String, Question)
|
|
176
|
+
}) {}
|
|
177
|
+
|
|
178
|
+
const noFailures: ReadonlyArray<QuestionFailure> = Object.freeze([])
|
|
179
|
+
|
|
180
|
+
export class JudgmentResult extends Schema.Class<JudgmentResult>("JudgmentResult")({
|
|
181
|
+
answers: Schema.Record(Schema.String, Answer),
|
|
182
|
+
failures: Schema.Array(QuestionFailure).pipe(
|
|
183
|
+
Schema.withConstructorDefault(Effect.succeed(noFailures))
|
|
184
|
+
),
|
|
185
|
+
usage: Schema.optionalKey(TokenUsage),
|
|
186
|
+
model: Schema.optionalKey(Schema.String),
|
|
187
|
+
backend: JudgmentBackend
|
|
188
|
+
}) {}
|
|
189
|
+
|
|
190
|
+
/** Confidence is the peak of the distribution: the maximum probability. */
|
|
191
|
+
export const confidenceOf = (probabilities: Readonly<Record<string, number>>): number =>
|
|
192
|
+
Object.values(probabilities).reduce((max, value) => (value > max ? value : max), 0)
|
|
193
|
+
|
|
194
|
+
/** Score is the expected level index under the distribution. */
|
|
195
|
+
export const expectedScore = (probabilities: Readonly<Record<string, number>>): number =>
|
|
196
|
+
Object.entries(probabilities).reduce((sum, [level, value]) => sum + Number(level) * value, 0)
|
|
197
|
+
|
|
198
|
+
export const argmax = (probabilities: Readonly<Record<string, number>>): string | undefined =>
|
|
199
|
+
Object.entries(probabilities).reduce<readonly [string, number] | undefined>(
|
|
200
|
+
(best, entry) => (best === undefined || entry[1] > best[1] ? entry : best),
|
|
201
|
+
undefined
|
|
202
|
+
)?.[0]
|
|
203
|
+
|
|
204
|
+
export const choiceAnswer = (
|
|
205
|
+
probabilities: Readonly<Record<string, number>>,
|
|
206
|
+
origin: AnswerOrigin,
|
|
207
|
+
support = 1
|
|
208
|
+
): ChoiceAnswer =>
|
|
209
|
+
ChoiceAnswer.make({
|
|
210
|
+
type: "choice",
|
|
211
|
+
choice: argmax(probabilities) ?? "",
|
|
212
|
+
probabilities,
|
|
213
|
+
confidence: confidenceOf(probabilities),
|
|
214
|
+
support,
|
|
215
|
+
origin
|
|
216
|
+
})
|
|
217
|
+
|
|
218
|
+
export const scoreAnswer = (
|
|
219
|
+
question: ScoreQuestion,
|
|
220
|
+
probabilities: Readonly<Record<string, number>>,
|
|
221
|
+
origin: AnswerOrigin,
|
|
222
|
+
support = 1
|
|
223
|
+
): ScoreAnswer =>
|
|
224
|
+
ScoreAnswer.make({
|
|
225
|
+
type: "score",
|
|
226
|
+
score: expectedScore(probabilities),
|
|
227
|
+
legend: Object.fromEntries(question.criteria.map((level, index) => [String(index), level])),
|
|
228
|
+
probabilities,
|
|
229
|
+
confidence: confidenceOf(probabilities),
|
|
230
|
+
support,
|
|
231
|
+
origin
|
|
232
|
+
})
|
|
233
|
+
|
|
234
|
+
export const truthAnswer = (probability: number, origin: AnswerOrigin, support = 1): TruthAnswer =>
|
|
235
|
+
TruthAnswer.make({ type: "truth", truth: probability, support, origin })
|
|
236
|
+
|
|
237
|
+
/** Origins for the common cases, so call sites stay short. */
|
|
238
|
+
export const origins = Object.freeze({
|
|
239
|
+
llm: (method: ScoringMethod, model?: string): AnswerOrigin =>
|
|
240
|
+
AnswerOrigin.make({ backend: "llm", method, ...(model === undefined ? {} : { model }) }),
|
|
241
|
+
hosted: (model?: string): AnswerOrigin =>
|
|
242
|
+
AnswerOrigin.make({
|
|
243
|
+
backend: "typesafe",
|
|
244
|
+
method: "hosted",
|
|
245
|
+
calibration: "claimed",
|
|
246
|
+
...(model === undefined ? {} : { model })
|
|
247
|
+
}),
|
|
248
|
+
fake: (method: ScoringMethod = "verbalized"): AnswerOrigin =>
|
|
249
|
+
AnswerOrigin.make({ backend: "fake", method }),
|
|
250
|
+
escalated: (backend: JudgmentBackend, model?: string): AnswerOrigin =>
|
|
251
|
+
AnswerOrigin.make({
|
|
252
|
+
backend,
|
|
253
|
+
method: "reasoning",
|
|
254
|
+
escalated: true,
|
|
255
|
+
...(model === undefined ? {} : { model })
|
|
256
|
+
})
|
|
257
|
+
})
|
|
258
|
+
|
|
259
|
+
/** Average several distributions over the same keys (used for permutations). */
|
|
260
|
+
export const averageProbabilities = (
|
|
261
|
+
distributions: ReadonlyArray<Readonly<Record<string, number>>>
|
|
262
|
+
): Record<string, number> => {
|
|
263
|
+
const keys = new Set(distributions.flatMap((distribution) => Object.keys(distribution)))
|
|
264
|
+
const count = Math.max(1, distributions.length)
|
|
265
|
+
return Object.fromEntries(
|
|
266
|
+
[...keys].map((key) => [
|
|
267
|
+
key,
|
|
268
|
+
distributions.reduce((sum, distribution) => sum + (distribution[key] ?? 0), 0) / count
|
|
269
|
+
])
|
|
270
|
+
)
|
|
271
|
+
}
|
|
272
|
+
|
|
273
|
+
export const renderState = (state: State): string =>
|
|
274
|
+
typeof state === "string"
|
|
275
|
+
? state
|
|
276
|
+
: Array.isArray(state)
|
|
277
|
+
? state.join("\n")
|
|
278
|
+
: JSON.stringify(state, null, 2)
|