@llm4ts/core 2.4.2 → 2.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (84) hide show
  1. package/dist/Connector.d.ts +4 -0
  2. package/dist/Connector.d.ts.map +1 -1
  3. package/dist/Connector.js +8 -3
  4. package/dist/Connector.js.map +1 -1
  5. package/dist/ConnectorConfig.d.ts.map +1 -1
  6. package/dist/ConnectorConfig.js +2 -0
  7. package/dist/ConnectorConfig.js.map +1 -1
  8. package/dist/ContextManagement.d.ts.map +1 -1
  9. package/dist/ContextManagement.js +1 -0
  10. package/dist/ContextManagement.js.map +1 -1
  11. package/dist/LabelScoring.d.ts +52 -0
  12. package/dist/LabelScoring.d.ts.map +1 -0
  13. package/dist/LabelScoring.js +139 -0
  14. package/dist/LabelScoring.js.map +1 -0
  15. package/dist/LlmService.d.ts +30 -2
  16. package/dist/LlmService.d.ts.map +1 -1
  17. package/dist/LlmService.js.map +1 -1
  18. package/dist/Models.d.ts +34 -2
  19. package/dist/Models.d.ts.map +1 -1
  20. package/dist/Models.js +37 -1
  21. package/dist/Models.js.map +1 -1
  22. package/dist/eval/Judge.d.ts +15 -0
  23. package/dist/eval/Judge.d.ts.map +1 -1
  24. package/dist/eval/Judge.js +50 -0
  25. package/dist/eval/Judge.js.map +1 -1
  26. package/dist/judgment/FakeJudgment.d.ts +22 -0
  27. package/dist/judgment/FakeJudgment.d.ts.map +1 -0
  28. package/dist/judgment/FakeJudgment.js +42 -0
  29. package/dist/judgment/FakeJudgment.js.map +1 -0
  30. package/dist/judgment/Judgment.d.ts +57 -0
  31. package/dist/judgment/Judgment.d.ts.map +1 -0
  32. package/dist/judgment/Judgment.js +57 -0
  33. package/dist/judgment/Judgment.js.map +1 -0
  34. package/dist/judgment/LlmJudgment.d.ts +73 -0
  35. package/dist/judgment/LlmJudgment.d.ts.map +1 -0
  36. package/dist/judgment/LlmJudgment.js +239 -0
  37. package/dist/judgment/LlmJudgment.js.map +1 -0
  38. package/dist/judgment/Schemas.d.ts +163 -0
  39. package/dist/judgment/Schemas.d.ts.map +1 -0
  40. package/dist/judgment/Schemas.js +197 -0
  41. package/dist/judgment/Schemas.js.map +1 -0
  42. package/dist/judgment/TypeSafeJudgment.d.ts +46 -0
  43. package/dist/judgment/TypeSafeJudgment.d.ts.map +1 -0
  44. package/dist/judgment/TypeSafeJudgment.js +185 -0
  45. package/dist/judgment/TypeSafeJudgment.js.map +1 -0
  46. package/dist/observability/MeteredLlmService.d.ts.map +1 -1
  47. package/dist/observability/MeteredLlmService.js +1 -0
  48. package/dist/observability/MeteredLlmService.js.map +1 -1
  49. package/dist/providers/ConnectorFactories.d.ts.map +1 -1
  50. package/dist/providers/ConnectorFactories.js +2 -0
  51. package/dist/providers/ConnectorFactories.js.map +1 -1
  52. package/dist/providers/LmStudioProvider.d.ts.map +1 -1
  53. package/dist/providers/LmStudioProvider.js +45 -34
  54. package/dist/providers/LmStudioProvider.js.map +1 -1
  55. package/dist/providers/MlxLmProvider.d.ts +29 -0
  56. package/dist/providers/MlxLmProvider.d.ts.map +1 -0
  57. package/dist/providers/MlxLmProvider.js +249 -0
  58. package/dist/providers/MlxLmProvider.js.map +1 -0
  59. package/dist/providers/MockProvider.d.ts.map +1 -1
  60. package/dist/providers/MockProvider.js +11 -0
  61. package/dist/providers/MockProvider.js.map +1 -1
  62. package/dist/providers/OpenAIModels.d.ts +35 -0
  63. package/dist/providers/OpenAIModels.d.ts.map +1 -1
  64. package/dist/providers/OpenAIModels.js +36 -3
  65. package/dist/providers/OpenAIModels.js.map +1 -1
  66. package/package.json +8 -1
  67. package/src/Connector.ts +13 -3
  68. package/src/ConnectorConfig.ts +2 -0
  69. package/src/ContextManagement.ts +1 -0
  70. package/src/LabelScoring.ts +211 -0
  71. package/src/LlmService.ts +37 -1
  72. package/src/Models.ts +43 -1
  73. package/src/eval/Judge.ts +74 -0
  74. package/src/judgment/FakeJudgment.ts +90 -0
  75. package/src/judgment/Judgment.ts +123 -0
  76. package/src/judgment/LlmJudgment.ts +399 -0
  77. package/src/judgment/Schemas.ts +278 -0
  78. package/src/judgment/TypeSafeJudgment.ts +253 -0
  79. package/src/observability/MeteredLlmService.ts +7 -0
  80. package/src/providers/ConnectorFactories.ts +4 -0
  81. package/src/providers/LmStudioProvider.ts +64 -48
  82. package/src/providers/MlxLmProvider.ts +357 -0
  83. package/src/providers/MockProvider.ts +19 -0
  84. package/src/providers/OpenAIModels.ts +40 -3
@@ -0,0 +1,399 @@
1
+ import * as Effect from "effect/Effect"
2
+ import * as Layer from "effect/Layer"
3
+ import * as Result from "effect/Result"
4
+ import * as Schema from "effect/Schema"
5
+ import type { LlmError } from "../Errors.ts"
6
+ import { verbalizedScoreLabelSequence, verbalizedScoreLabels } from "../LabelScoring.ts"
7
+ import type { LabelSequence, LlmServiceShape } from "../LlmService.ts"
8
+ import { TokenUsage, type LabelDistribution } from "../Models.ts"
9
+ import { Judgment, type JudgmentInput, type JudgmentShape } from "./Judgment.ts"
10
+ import {
11
+ averageProbabilities,
12
+ choiceAnswer,
13
+ JudgmentResult,
14
+ QuestionFailure,
15
+ renderState,
16
+ scoreAnswer,
17
+ truthAnswer,
18
+ origins,
19
+ type Answer,
20
+ type AnswerOrigin,
21
+ type Description,
22
+ type Question,
23
+ type ScoringMethod,
24
+ type State
25
+ } from "./Schemas.ts"
26
+
27
+ /**
28
+ * `LlmJudgment`: the Judgment service over any `LlmServiceShape`. One label
29
+ * question per atomic question, answered through `scoreLabels`: a single
30
+ * forward pass on connectors with log-probabilities, a short JSON reply on
31
+ * the rest. Questions never see each other's answers in `independent`
32
+ * batching; `shared-prefix` sends the state once with every question and
33
+ * reads each answer as its own label distribution, trading that isolation
34
+ * for one prompt prefix per request.
35
+ */
36
+
37
+ export const Batching = Schema.Literals(["independent", "shared-prefix"])
38
+ export type Batching = typeof Batching.Type
39
+
40
+ export class LlmJudgmentConfig extends Schema.Class<LlmJudgmentConfig>("LlmJudgmentConfig")({
41
+ /** Questions in flight at once. 1 keeps a local server's prompt cache warm. */
42
+ concurrency: Schema.Int.pipe(Schema.withConstructorDefault(Effect.succeed(1))),
43
+ /** 2 also asks with the options reversed and averages, to blunt position bias. */
44
+ permutations: Schema.Literals([1, 2]).pipe(
45
+ Schema.withConstructorDefault(Effect.succeed<1 | 2>(1))
46
+ ),
47
+ /**
48
+ * `independent` (default): one call per question, Jev's answer
49
+ * independence. `shared-prefix`: every question of a request in one call
50
+ * whose state prefix is sent once; a question the batch could not read
51
+ * falls back to its own independent call.
52
+ */
53
+ batching: Batching.pipe(Schema.withConstructorDefault(Effect.succeed<Batching>("independent"))),
54
+ /** Connector and model behind the seat, for `identity` and answer origins. */
55
+ connector: Schema.optionalKey(Schema.String),
56
+ model: Schema.optionalKey(Schema.String)
57
+ }) {}
58
+
59
+ export interface LlmJudgmentHooks {
60
+ /** Called once per request with the summed backend usage, when any was reported. */
61
+ readonly onUsage?: (usage: TokenUsage, model: string | undefined) => Effect.Effect<void>
62
+ /** Called when a question fell back from label scoring to the verbalized path. */
63
+ readonly onFallback?: (key: string, reason: string) => Effect.Effect<void>
64
+ }
65
+
66
+ const LABELS = "ABCDEFGHIJKLMNOPQRSTUVWXYZ"
67
+
68
+ const describe = (description: Description): string =>
69
+ typeof description === "string" ? description : JSON.stringify(description)
70
+
71
+ interface LabelPlan {
72
+ readonly prompt: string
73
+ readonly labels: ReadonlyArray<string>
74
+ /** Label to the option key or level index it stands for. */
75
+ readonly keys: Readonly<Record<string, string>>
76
+ }
77
+
78
+ /** A question's part of a prompt, without the state header. */
79
+ export interface QuestionBody {
80
+ readonly body: string
81
+ readonly labels: ReadonlyArray<string>
82
+ readonly keys: Readonly<Record<string, string>>
83
+ }
84
+
85
+ export const questionBody = (question: Question, reverse: boolean): QuestionBody => {
86
+ if (question.type === "truth") {
87
+ const criteria =
88
+ question.criteria === undefined
89
+ ? ""
90
+ : `\nYes means: ${describe(question.criteria.true)}\nNo means: ${describe(question.criteria.false)}`
91
+ return {
92
+ body: `Statement: ${question.instructions}${criteria}\nAnswer with exactly one word: yes or no.`,
93
+ labels: ["yes", "no"],
94
+ keys: { yes: "yes", no: "no" }
95
+ }
96
+ }
97
+ const entries: ReadonlyArray<readonly [key: string, description: Description]> =
98
+ question.type === "choice"
99
+ ? Object.entries(question.criteria)
100
+ : question.criteria.map((level, index) => [String(index), level] as const)
101
+ const ordered = reverse ? [...entries].reverse() : entries
102
+ const labels = ordered.map((_, index) => LABELS[index] ?? `L${index}`)
103
+ const lines = ordered
104
+ .map(([key, description], index) =>
105
+ question.type === "choice"
106
+ ? `${labels[index]}. ${key}: ${describe(description)}`
107
+ : `${labels[index]}. ${describe(description)}`
108
+ )
109
+ .join("\n")
110
+ const ask = question.type === "choice" ? "Question" : "Rate on the levels below"
111
+ return {
112
+ body: `${ask}: ${question.instructions}\nOptions:\n${lines}\nAnswer with exactly one letter.`,
113
+ labels,
114
+ keys: Object.fromEntries(ordered.map(([key], index) => [labels[index] ?? `L${index}`, key]))
115
+ }
116
+ }
117
+
118
+ const stateHeader = (state: State): string => `State:\n${renderState(state)}\n\n`
119
+
120
+ /**
121
+ * The fixed label-scoring template: state first, then the question, then
122
+ * one single-token label per option. Truth uses `yes`/`no`.
123
+ */
124
+ export const labelPlan = (state: State, question: Question, reverse: boolean): LabelPlan => {
125
+ const part = questionBody(question, reverse)
126
+ return { prompt: `${stateHeader(state)}${part.body}`, labels: part.labels, keys: part.keys }
127
+ }
128
+
129
+ /**
130
+ * The shared-prefix template: the state once, then every question numbered,
131
+ * and an answer format of one label per line so a backend can read each
132
+ * position as its own distribution.
133
+ */
134
+ export const sequencePlan = (
135
+ state: State,
136
+ questions: ReadonlyArray<Question>,
137
+ reverse: boolean
138
+ ): { readonly prompt: string; readonly parts: ReadonlyArray<QuestionBody> } => {
139
+ const parts = questions.map((question) => questionBody(question, reverse))
140
+ const numbered = parts.map((part, index) => `Question ${index + 1}.\n${part.body}`).join("\n\n")
141
+ return {
142
+ prompt:
143
+ `${stateHeader(state)}Answer each numbered question below with exactly one label on its own line, ` +
144
+ `formatted as "<number>: <label>", in order, and nothing else.\n\n${numbered}`,
145
+ parts
146
+ }
147
+ }
148
+
149
+ const byKey = (
150
+ keys: Readonly<Record<string, string>>,
151
+ distribution: LabelDistribution
152
+ ): Record<string, number> =>
153
+ Object.fromEntries(
154
+ Object.entries(distribution.probabilities).map(([label, value]) => [
155
+ keys[label] ?? label,
156
+ value
157
+ ])
158
+ )
159
+
160
+ const toAnswer = (
161
+ question: Question,
162
+ probabilities: Readonly<Record<string, number>>,
163
+ origin: AnswerOrigin,
164
+ support: number
165
+ ): Answer =>
166
+ question.type === "choice"
167
+ ? choiceAnswer(probabilities, origin, support)
168
+ : question.type === "score"
169
+ ? scoreAnswer(question, probabilities, origin, support)
170
+ : truthAnswer(probabilities["yes"] ?? 0, origin, support)
171
+
172
+ const sumUsage = (usages: ReadonlyArray<TokenUsage | undefined>): TokenUsage | undefined => {
173
+ const present = usages.filter((usage): usage is TokenUsage => usage !== undefined)
174
+ return present.length === 0
175
+ ? undefined
176
+ : TokenUsage.make({
177
+ prompt: present.reduce((sum, usage) => sum + usage.prompt, 0),
178
+ completion: present.reduce((sum, usage) => sum + usage.completion, 0),
179
+ total: present.reduce((sum, usage) => sum + usage.total, 0)
180
+ })
181
+ }
182
+
183
+ interface Scored {
184
+ readonly answer: Answer
185
+ readonly usage: TokenUsage | undefined
186
+ readonly model: string | undefined
187
+ }
188
+
189
+ type Read = readonly [keys: Readonly<Record<string, string>>, distribution: LabelDistribution]
190
+
191
+ /** Combine one distribution per permutation into an answer; the weaker method and support win. */
192
+ const combine = (
193
+ question: Question,
194
+ reads: ReadonlyArray<Read>,
195
+ fallbackModel: string | undefined
196
+ ): Scored => {
197
+ const method: ScoringMethod = reads.every(
198
+ ([, distribution]) => distribution.method === "logprobs"
199
+ )
200
+ ? "logprobs"
201
+ : reads.some(([, distribution]) => distribution.method === "sampled")
202
+ ? "sampled"
203
+ : "verbalized"
204
+ const support = Math.min(...reads.map(([, distribution]) => distribution.support))
205
+ const probabilities = averageProbabilities(
206
+ reads.map(([keys, distribution]) => byKey(keys, distribution))
207
+ )
208
+ const model =
209
+ reads.find(([, distribution]) => distribution.model !== undefined)?.[1].model ?? fallbackModel
210
+ return {
211
+ answer: toAnswer(question, probabilities, origins.llm(method, model), support),
212
+ usage: sumUsage(reads.map(([, distribution]) => distribution.usage)),
213
+ model
214
+ }
215
+ }
216
+
217
+ type Outcome = readonly [key: string, outcome: Result.Result<Scored, LlmError>]
218
+
219
+ export const makeLlmJudgment = (
220
+ llm: LlmServiceShape,
221
+ config: LlmJudgmentConfig = LlmJudgmentConfig.make({}),
222
+ hooks: LlmJudgmentHooks = {}
223
+ ): JudgmentShape => {
224
+ const fallback = verbalizedScoreLabels(llm.executeStructuredWithUsage)
225
+ const sequence =
226
+ llm.scoreLabelSequence ?? verbalizedScoreLabelSequence(llm.executeStructuredWithUsage)
227
+ const permutations: ReadonlyArray<boolean> = config.permutations === 2 ? [false, true] : [false]
228
+ const permutationsFor = (question: Question): ReadonlyArray<boolean> =>
229
+ permutations.filter((reverse) => !reverse || question.type !== "truth")
230
+
231
+ const scoreOnce = (plan: LabelPlan, key: string): Effect.Effect<LabelDistribution, LlmError> =>
232
+ llm
233
+ .scoreLabels(plan.prompt, plan.labels)
234
+ .pipe(
235
+ Effect.catchTag("ParseError", (error) =>
236
+ (hooks.onFallback?.(key, error.message) ?? Effect.void).pipe(
237
+ Effect.andThen(fallback(plan.prompt, plan.labels))
238
+ )
239
+ )
240
+ )
241
+
242
+ const scoreQuestion = (
243
+ state: State,
244
+ key: string,
245
+ question: Question
246
+ ): Effect.Effect<Scored, LlmError> =>
247
+ Effect.gen(function* () {
248
+ const plans = permutationsFor(question).map((reverse) => labelPlan(state, question, reverse))
249
+ const reads = yield* Effect.forEach(plans, (plan) =>
250
+ Effect.map(scoreOnce(plan, key), (distribution): Read => [plan.keys, distribution])
251
+ )
252
+ return combine(question, reads, config.model)
253
+ })
254
+
255
+ const independently = (
256
+ input: JudgmentInput,
257
+ entries: ReadonlyArray<readonly [string, Question]>
258
+ ): Effect.Effect<ReadonlyArray<Outcome>> =>
259
+ Effect.forEach(
260
+ entries,
261
+ ([key, question]) =>
262
+ Effect.map(
263
+ Effect.result(scoreQuestion(input.state, key, question)),
264
+ (outcome): Outcome => [key, outcome]
265
+ ),
266
+ { concurrency: config.concurrency }
267
+ )
268
+
269
+ /**
270
+ * One call per permutation for the whole request. A question whose
271
+ * position could not be read in any permutation, or a call that failed
272
+ * outright, goes back through the independent path for that question.
273
+ */
274
+ const sharedPrefix = (
275
+ input: JudgmentInput,
276
+ entries: ReadonlyArray<readonly [string, Question]>
277
+ ): Effect.Effect<{
278
+ readonly outcomes: ReadonlyArray<Outcome>
279
+ readonly usage: TokenUsage | undefined
280
+ readonly model: string | undefined
281
+ }> =>
282
+ Effect.gen(function* () {
283
+ const questions = entries.map(([, question]) => question)
284
+ const calls = yield* Effect.forEach(permutations, (reverse) => {
285
+ const plan = sequencePlan(input.state, questions, reverse)
286
+ return Effect.map(
287
+ Effect.result(
288
+ sequence(
289
+ plan.prompt,
290
+ plan.parts.map((part) => part.labels)
291
+ )
292
+ ),
293
+ (outcome) => [plan, outcome] as const
294
+ )
295
+ })
296
+ const successes = calls.flatMap(([, outcome]) =>
297
+ Result.isSuccess(outcome) ? [outcome.success] : []
298
+ )
299
+ const batchModel = successes.find((call) => call.model !== undefined)?.model ?? config.model
300
+ const answered: Array<Outcome> = []
301
+ const retry: Array<readonly [string, Question]> = []
302
+ for (const [index, [key, question]] of entries.entries()) {
303
+ const reads: Array<Read> = []
304
+ let reason: string | undefined
305
+ for (const [callIndex, reverse] of permutations.entries()) {
306
+ if (!permutationsFor(question).includes(reverse)) {
307
+ continue
308
+ }
309
+ const call = calls[callIndex]
310
+ const part = call?.[0].parts[index]
311
+ const outcome = call?.[1]
312
+ if (outcome === undefined || part === undefined) {
313
+ reason = "no batched call covered this question"
314
+ break
315
+ }
316
+ if (Result.isFailure(outcome)) {
317
+ reason = outcome.failure.message
318
+ break
319
+ }
320
+ const entry = outcome.success.entries[index]
321
+ if (entry === undefined) {
322
+ reason = "no entry for this question"
323
+ break
324
+ }
325
+ if (Result.isFailure(entry)) {
326
+ reason = entry.failure.message
327
+ break
328
+ }
329
+ reads.push([part.keys, entry.success])
330
+ }
331
+ if (reason === undefined && reads.length > 0) {
332
+ // Usage was reported once for the whole call; the per-question reads carry none.
333
+ answered.push([key, Result.succeed(combine(question, reads, batchModel))])
334
+ } else {
335
+ yield* hooks.onFallback?.(key, `shared-prefix batch: ${reason ?? "unreadable"}`) ??
336
+ Effect.void
337
+ retry.push([key, question])
338
+ }
339
+ }
340
+ const retried = retry.length === 0 ? [] : yield* independently(input, retry)
341
+ const order = new Map(entries.map(([key], index) => [key, index] as const))
342
+ const outcomes = [...answered, ...retried].sort(
343
+ ([a], [b]) => (order.get(a) ?? 0) - (order.get(b) ?? 0)
344
+ )
345
+ return {
346
+ outcomes,
347
+ usage: sumUsage(successes.map((call) => call.usage)),
348
+ model: successes.find((call) => call.model !== undefined)?.model
349
+ }
350
+ })
351
+
352
+ const judge = Effect.fn("@llm4ts/core/judgment/LlmJudgment.judge")(function* (
353
+ input: JudgmentInput
354
+ ): Effect.fn.Return<JudgmentResult> {
355
+ const entries = Object.entries(input.questions)
356
+ const batched =
357
+ config.batching === "shared-prefix" && entries.length > 1
358
+ ? yield* sharedPrefix(input, entries)
359
+ : { outcomes: yield* independently(input, entries), usage: undefined, model: undefined }
360
+ const answers: Record<string, Answer> = {}
361
+ const failures: Array<QuestionFailure> = []
362
+ const usages: Array<TokenUsage | undefined> = [batched.usage]
363
+ let model: string | undefined = batched.model
364
+ for (const [key, outcome] of batched.outcomes) {
365
+ if (Result.isFailure(outcome)) {
366
+ failures.push(QuestionFailure.make({ key, reason: outcome.failure.message }))
367
+ } else {
368
+ answers[key] = outcome.success.answer
369
+ usages.push(outcome.success.usage)
370
+ model = model ?? outcome.success.model
371
+ }
372
+ }
373
+ const usage = sumUsage(usages)
374
+ if (usage !== undefined && hooks.onUsage !== undefined) {
375
+ yield* hooks.onUsage(usage, model)
376
+ }
377
+ return JudgmentResult.make({
378
+ answers,
379
+ failures,
380
+ backend: "llm",
381
+ ...(usage === undefined ? {} : { usage }),
382
+ ...(model === undefined ? {} : { model })
383
+ })
384
+ })
385
+
386
+ return {
387
+ backend: "llm",
388
+ identity: `llm:${config.connector ?? "unknown"}:${config.model ?? "default"}`,
389
+ judge
390
+ }
391
+ }
392
+
393
+ export const LlmJudgmentLive = (
394
+ llm: LlmServiceShape,
395
+ config?: LlmJudgmentConfig,
396
+ hooks?: LlmJudgmentHooks
397
+ ): Layer.Layer<Judgment> => Layer.succeed(Judgment, makeLlmJudgment(llm, config, hooks))
398
+
399
+ export type { LabelSequence }
@@ -0,0 +1,278 @@
1
+ import * as Effect from "effect/Effect"
2
+ import * as Schema from "effect/Schema"
3
+ import { TokenUsage } from "../Models.ts"
4
+
5
+ /**
6
+ * Typed judgments (ADR 0017): atomic questions evaluated against one State,
7
+ * answered with probabilities instead of text. Wire names mirror TypeSafe's
8
+ * Jev so its cookbooks port by search and replace, with two changes: `noul`
9
+ * is `truth`, and every answer carries an `origin` (backend, checkpoint,
10
+ * extraction method, calibration evidence, escalation) plus `support`, the
11
+ * mass the backend actually placed on the offered options. Vocabulary:
12
+ * CONTEXT.md.
13
+ */
14
+
15
+ /** The content a judgment evaluates. Data, never instructions. */
16
+ export const State = Schema.Union([
17
+ Schema.String,
18
+ Schema.Record(Schema.String, Schema.Json),
19
+ Schema.Array(Schema.String)
20
+ ])
21
+ export type State = typeof State.Type
22
+
23
+ /** An option, level, or criterion: text or any JSON structure (Jev's `EntryType`). */
24
+ export const Description = Schema.Json
25
+ export type Description = typeof Description.Type
26
+
27
+ export class ChoiceQuestion extends Schema.Class<ChoiceQuestion>("ChoiceQuestion")({
28
+ type: Schema.Literal("choice"),
29
+ instructions: Schema.String,
30
+ /** Option key to description. Include an "other" key when coverage is uncertain. */
31
+ criteria: Schema.Record(Schema.String, Description)
32
+ }) {}
33
+
34
+ export class ScoreQuestion extends Schema.Class<ScoreQuestion>("ScoreQuestion")({
35
+ type: Schema.Literal("score"),
36
+ instructions: Schema.String,
37
+ /** Ordered levels, index 0 first. */
38
+ criteria: Schema.Array(Description)
39
+ }) {}
40
+
41
+ export class TruthCriteria extends Schema.Class<TruthCriteria>("TruthCriteria")({
42
+ true: Description,
43
+ false: Description
44
+ }) {}
45
+
46
+ export class TruthQuestion extends Schema.Class<TruthQuestion>("TruthQuestion")({
47
+ type: Schema.Literal("truth"),
48
+ instructions: Schema.String,
49
+ criteria: Schema.optionalKey(TruthCriteria)
50
+ }) {}
51
+
52
+ export const Question = Schema.Union([ChoiceQuestion, ScoreQuestion, TruthQuestion])
53
+ export type Question = typeof Question.Type
54
+
55
+ export const choice = (
56
+ instructions: string,
57
+ criteria: Readonly<Record<string, Description>>
58
+ ): ChoiceQuestion => ChoiceQuestion.make({ type: "choice", instructions, criteria })
59
+
60
+ export const score = (instructions: string, criteria: ReadonlyArray<Description>): ScoreQuestion =>
61
+ ScoreQuestion.make({ type: "score", instructions, criteria })
62
+
63
+ export const truth = (
64
+ instructions: string,
65
+ criteria?: { readonly true: Description; readonly false: Description }
66
+ ): TruthQuestion =>
67
+ TruthQuestion.make({
68
+ type: "truth",
69
+ instructions,
70
+ ...(criteria === undefined ? {} : { criteria: TruthCriteria.make(criteria) })
71
+ })
72
+
73
+ export const JudgmentBackend = Schema.Literals(["typesafe", "llm", "fake"])
74
+ export type JudgmentBackend = typeof JudgmentBackend.Type
75
+
76
+ /** How the probabilities were extracted. */
77
+ export const ScoringMethod = Schema.Literals([
78
+ "logprobs",
79
+ "verbalized",
80
+ "sampled",
81
+ "reasoning",
82
+ "hosted"
83
+ ])
84
+ export type ScoringMethod = typeof ScoringMethod.Type
85
+
86
+ /**
87
+ * What is known about the numbers' calibration: `none` (nothing), `claimed`
88
+ * (the provider says so; TypeSafe), `measured` (an evaluation in this
89
+ * project produced the evidence; see the judgment decision map).
90
+ */
91
+ export const Calibration = Schema.Literals(["none", "claimed", "measured"])
92
+ export type Calibration = typeof Calibration.Type
93
+
94
+ /**
95
+ * Where an answer came from, in four separate facts a policy may key on:
96
+ * which backend and checkpoint produced it, how the probabilities were
97
+ * extracted, what is known about their calibration, and whether the answer
98
+ * replaced an earlier one through escalation.
99
+ */
100
+ export class AnswerOrigin extends Schema.Class<AnswerOrigin>("AnswerOrigin")({
101
+ backend: JudgmentBackend,
102
+ model: Schema.optionalKey(Schema.String),
103
+ method: ScoringMethod,
104
+ calibration: Calibration.pipe(
105
+ Schema.withConstructorDefault(Effect.succeed<Calibration>("none")),
106
+ Schema.withDecodingDefaultKey(Effect.succeed<Calibration>("none"))
107
+ ),
108
+ escalated: Schema.Boolean.pipe(
109
+ Schema.withConstructorDefault(Effect.succeed(false)),
110
+ Schema.withDecodingDefaultKey(Effect.succeed(false))
111
+ )
112
+ }) {}
113
+
114
+ /**
115
+ * Probability mass the backend placed on the offered options before
116
+ * renormalization (1 when declared over the options alone). Low support
117
+ * means the distribution was rebuilt from a sliver and must not be acted on.
118
+ */
119
+ const Support = Schema.Number.pipe(
120
+ Schema.withConstructorDefault(Effect.succeed(1)),
121
+ Schema.withDecodingDefaultKey(Effect.succeed(1))
122
+ )
123
+
124
+ export class ChoiceAnswer extends Schema.Class<ChoiceAnswer>("ChoiceAnswer")({
125
+ type: Schema.Literal("choice"),
126
+ choice: Schema.String,
127
+ probabilities: Schema.Record(Schema.String, Schema.Number),
128
+ /** The maximum probability, everywhere: the same statistic whatever the backend. */
129
+ confidence: Schema.Number,
130
+ /** The backend's own confidence statistic when it reports one (TypeSafe), kept for comparison. */
131
+ reportedConfidence: Schema.optionalKey(Schema.Number),
132
+ support: Support,
133
+ origin: AnswerOrigin
134
+ }) {}
135
+
136
+ export class ScoreAnswer extends Schema.Class<ScoreAnswer>("ScoreAnswer")({
137
+ type: Schema.Literal("score"),
138
+ /** Expected level index; may fall between two levels. */
139
+ score: Schema.Number,
140
+ /** Level index (as a string key) to its description, echoed from the question. */
141
+ legend: Schema.Record(Schema.String, Description),
142
+ probabilities: Schema.Record(Schema.String, Schema.Number),
143
+ confidence: Schema.Number,
144
+ reportedConfidence: Schema.optionalKey(Schema.Number),
145
+ support: Support,
146
+ origin: AnswerOrigin
147
+ }) {}
148
+
149
+ export class TruthAnswer extends Schema.Class<TruthAnswer>("TruthAnswer")({
150
+ type: Schema.Literal("truth"),
151
+ /** Probability that the statement holds. */
152
+ truth: Schema.Number,
153
+ support: Support,
154
+ origin: AnswerOrigin
155
+ }) {}
156
+
157
+ export const Answer = Schema.Union([ChoiceAnswer, ScoreAnswer, TruthAnswer])
158
+ export type Answer = typeof Answer.Type
159
+
160
+ /** The answer type a question kind produces, for typed call sites. */
161
+ export type AnswerFor<Q extends Question> = Q extends ChoiceQuestion
162
+ ? ChoiceAnswer
163
+ : Q extends ScoreQuestion
164
+ ? ScoreAnswer
165
+ : TruthAnswer
166
+
167
+ /** A question that could not be answered; the others in the request still were. */
168
+ export class QuestionFailure extends Schema.Class<QuestionFailure>("QuestionFailure")({
169
+ key: Schema.String,
170
+ reason: Schema.String
171
+ }) {}
172
+
173
+ export class JudgmentRequest extends Schema.Class<JudgmentRequest>("JudgmentRequest")({
174
+ state: State,
175
+ questions: Schema.Record(Schema.String, Question)
176
+ }) {}
177
+
178
+ const noFailures: ReadonlyArray<QuestionFailure> = Object.freeze([])
179
+
180
+ export class JudgmentResult extends Schema.Class<JudgmentResult>("JudgmentResult")({
181
+ answers: Schema.Record(Schema.String, Answer),
182
+ failures: Schema.Array(QuestionFailure).pipe(
183
+ Schema.withConstructorDefault(Effect.succeed(noFailures))
184
+ ),
185
+ usage: Schema.optionalKey(TokenUsage),
186
+ model: Schema.optionalKey(Schema.String),
187
+ backend: JudgmentBackend
188
+ }) {}
189
+
190
+ /** Confidence is the peak of the distribution: the maximum probability. */
191
+ export const confidenceOf = (probabilities: Readonly<Record<string, number>>): number =>
192
+ Object.values(probabilities).reduce((max, value) => (value > max ? value : max), 0)
193
+
194
+ /** Score is the expected level index under the distribution. */
195
+ export const expectedScore = (probabilities: Readonly<Record<string, number>>): number =>
196
+ Object.entries(probabilities).reduce((sum, [level, value]) => sum + Number(level) * value, 0)
197
+
198
+ export const argmax = (probabilities: Readonly<Record<string, number>>): string | undefined =>
199
+ Object.entries(probabilities).reduce<readonly [string, number] | undefined>(
200
+ (best, entry) => (best === undefined || entry[1] > best[1] ? entry : best),
201
+ undefined
202
+ )?.[0]
203
+
204
+ export const choiceAnswer = (
205
+ probabilities: Readonly<Record<string, number>>,
206
+ origin: AnswerOrigin,
207
+ support = 1
208
+ ): ChoiceAnswer =>
209
+ ChoiceAnswer.make({
210
+ type: "choice",
211
+ choice: argmax(probabilities) ?? "",
212
+ probabilities,
213
+ confidence: confidenceOf(probabilities),
214
+ support,
215
+ origin
216
+ })
217
+
218
+ export const scoreAnswer = (
219
+ question: ScoreQuestion,
220
+ probabilities: Readonly<Record<string, number>>,
221
+ origin: AnswerOrigin,
222
+ support = 1
223
+ ): ScoreAnswer =>
224
+ ScoreAnswer.make({
225
+ type: "score",
226
+ score: expectedScore(probabilities),
227
+ legend: Object.fromEntries(question.criteria.map((level, index) => [String(index), level])),
228
+ probabilities,
229
+ confidence: confidenceOf(probabilities),
230
+ support,
231
+ origin
232
+ })
233
+
234
+ export const truthAnswer = (probability: number, origin: AnswerOrigin, support = 1): TruthAnswer =>
235
+ TruthAnswer.make({ type: "truth", truth: probability, support, origin })
236
+
237
+ /** Origins for the common cases, so call sites stay short. */
238
+ export const origins = Object.freeze({
239
+ llm: (method: ScoringMethod, model?: string): AnswerOrigin =>
240
+ AnswerOrigin.make({ backend: "llm", method, ...(model === undefined ? {} : { model }) }),
241
+ hosted: (model?: string): AnswerOrigin =>
242
+ AnswerOrigin.make({
243
+ backend: "typesafe",
244
+ method: "hosted",
245
+ calibration: "claimed",
246
+ ...(model === undefined ? {} : { model })
247
+ }),
248
+ fake: (method: ScoringMethod = "verbalized"): AnswerOrigin =>
249
+ AnswerOrigin.make({ backend: "fake", method }),
250
+ escalated: (backend: JudgmentBackend, model?: string): AnswerOrigin =>
251
+ AnswerOrigin.make({
252
+ backend,
253
+ method: "reasoning",
254
+ escalated: true,
255
+ ...(model === undefined ? {} : { model })
256
+ })
257
+ })
258
+
259
+ /** Average several distributions over the same keys (used for permutations). */
260
+ export const averageProbabilities = (
261
+ distributions: ReadonlyArray<Readonly<Record<string, number>>>
262
+ ): Record<string, number> => {
263
+ const keys = new Set(distributions.flatMap((distribution) => Object.keys(distribution)))
264
+ const count = Math.max(1, distributions.length)
265
+ return Object.fromEntries(
266
+ [...keys].map((key) => [
267
+ key,
268
+ distributions.reduce((sum, distribution) => sum + (distribution[key] ?? 0), 0) / count
269
+ ])
270
+ )
271
+ }
272
+
273
+ export const renderState = (state: State): string =>
274
+ typeof state === "string"
275
+ ? state
276
+ : Array.isArray(state)
277
+ ? state.join("\n")
278
+ : JSON.stringify(state, null, 2)