@llm4ts/core 2.5.0 → 2.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (84) hide show
  1. package/dist/Connector.d.ts +4 -0
  2. package/dist/Connector.d.ts.map +1 -1
  3. package/dist/Connector.js +8 -3
  4. package/dist/Connector.js.map +1 -1
  5. package/dist/ConnectorConfig.d.ts.map +1 -1
  6. package/dist/ConnectorConfig.js +2 -0
  7. package/dist/ConnectorConfig.js.map +1 -1
  8. package/dist/ContextManagement.d.ts.map +1 -1
  9. package/dist/ContextManagement.js +1 -0
  10. package/dist/ContextManagement.js.map +1 -1
  11. package/dist/LabelScoring.d.ts +52 -0
  12. package/dist/LabelScoring.d.ts.map +1 -0
  13. package/dist/LabelScoring.js +139 -0
  14. package/dist/LabelScoring.js.map +1 -0
  15. package/dist/LlmService.d.ts +30 -2
  16. package/dist/LlmService.d.ts.map +1 -1
  17. package/dist/LlmService.js.map +1 -1
  18. package/dist/Models.d.ts +34 -2
  19. package/dist/Models.d.ts.map +1 -1
  20. package/dist/Models.js +37 -1
  21. package/dist/Models.js.map +1 -1
  22. package/dist/eval/Judge.d.ts +15 -0
  23. package/dist/eval/Judge.d.ts.map +1 -1
  24. package/dist/eval/Judge.js +50 -0
  25. package/dist/eval/Judge.js.map +1 -1
  26. package/dist/judgment/FakeJudgment.d.ts +22 -0
  27. package/dist/judgment/FakeJudgment.d.ts.map +1 -0
  28. package/dist/judgment/FakeJudgment.js +42 -0
  29. package/dist/judgment/FakeJudgment.js.map +1 -0
  30. package/dist/judgment/Judgment.d.ts +57 -0
  31. package/dist/judgment/Judgment.d.ts.map +1 -0
  32. package/dist/judgment/Judgment.js +57 -0
  33. package/dist/judgment/Judgment.js.map +1 -0
  34. package/dist/judgment/LlmJudgment.d.ts +73 -0
  35. package/dist/judgment/LlmJudgment.d.ts.map +1 -0
  36. package/dist/judgment/LlmJudgment.js +239 -0
  37. package/dist/judgment/LlmJudgment.js.map +1 -0
  38. package/dist/judgment/Schemas.d.ts +163 -0
  39. package/dist/judgment/Schemas.d.ts.map +1 -0
  40. package/dist/judgment/Schemas.js +197 -0
  41. package/dist/judgment/Schemas.js.map +1 -0
  42. package/dist/judgment/TypeSafeJudgment.d.ts +46 -0
  43. package/dist/judgment/TypeSafeJudgment.d.ts.map +1 -0
  44. package/dist/judgment/TypeSafeJudgment.js +185 -0
  45. package/dist/judgment/TypeSafeJudgment.js.map +1 -0
  46. package/dist/observability/MeteredLlmService.d.ts.map +1 -1
  47. package/dist/observability/MeteredLlmService.js +1 -0
  48. package/dist/observability/MeteredLlmService.js.map +1 -1
  49. package/dist/providers/ConnectorFactories.d.ts.map +1 -1
  50. package/dist/providers/ConnectorFactories.js +2 -0
  51. package/dist/providers/ConnectorFactories.js.map +1 -1
  52. package/dist/providers/LmStudioProvider.d.ts.map +1 -1
  53. package/dist/providers/LmStudioProvider.js +45 -34
  54. package/dist/providers/LmStudioProvider.js.map +1 -1
  55. package/dist/providers/MlxLmProvider.d.ts +29 -0
  56. package/dist/providers/MlxLmProvider.d.ts.map +1 -0
  57. package/dist/providers/MlxLmProvider.js +249 -0
  58. package/dist/providers/MlxLmProvider.js.map +1 -0
  59. package/dist/providers/MockProvider.d.ts.map +1 -1
  60. package/dist/providers/MockProvider.js +11 -0
  61. package/dist/providers/MockProvider.js.map +1 -1
  62. package/dist/providers/OpenAIModels.d.ts +35 -0
  63. package/dist/providers/OpenAIModels.d.ts.map +1 -1
  64. package/dist/providers/OpenAIModels.js +36 -3
  65. package/dist/providers/OpenAIModels.js.map +1 -1
  66. package/package.json +8 -1
  67. package/src/Connector.ts +13 -3
  68. package/src/ConnectorConfig.ts +2 -0
  69. package/src/ContextManagement.ts +1 -0
  70. package/src/LabelScoring.ts +211 -0
  71. package/src/LlmService.ts +37 -1
  72. package/src/Models.ts +43 -1
  73. package/src/eval/Judge.ts +74 -0
  74. package/src/judgment/FakeJudgment.ts +90 -0
  75. package/src/judgment/Judgment.ts +123 -0
  76. package/src/judgment/LlmJudgment.ts +399 -0
  77. package/src/judgment/Schemas.ts +278 -0
  78. package/src/judgment/TypeSafeJudgment.ts +253 -0
  79. package/src/observability/MeteredLlmService.ts +7 -0
  80. package/src/providers/ConnectorFactories.ts +4 -0
  81. package/src/providers/LmStudioProvider.ts +64 -48
  82. package/src/providers/MlxLmProvider.ts +357 -0
  83. package/src/providers/MockProvider.ts +19 -0
  84. package/src/providers/OpenAIModels.ts +40 -3
@@ -0,0 +1,211 @@
1
+ import * as Effect from "effect/Effect"
2
+ import * as Schema from "effect/Schema"
3
+ import { InvalidRequestError, ParseError, type LlmError } from "./Errors.ts"
4
+ import * as Result from "effect/Result"
5
+ import type { LabelSequence, LlmServiceShape } from "./LlmService.ts"
6
+ import { LabelDistribution, type JsonSchema, type LabelMethod } from "./Models.ts"
7
+
8
+ /**
9
+ * Label scoring: the one classification primitive every connector offers.
10
+ *
11
+ * A caller hands over a prompt and a closed set of labels and gets a
12
+ * probability per label back. Connectors that expose token log-probabilities
13
+ * implement it natively in one forward pass; everything else derives it here
14
+ * from `executeStructured`, asking the model to write the numbers down
15
+ * (method `verbalized`, which callers should hold to a higher bar).
16
+ */
17
+
18
+ const raw = (text: string): string => (text.length > 200 ? `${text.slice(0, 200)}…` : text)
19
+
20
+ /**
21
+ * Keep only the offered labels, clamp negatives to zero and renormalize.
22
+ * `support` records how much mass the offered labels held before
23
+ * renormalization: for log-probabilities that is their absolute share of
24
+ * the vocabulary distribution (so one label found with 5% of the mass
25
+ * yields probability 1 but support 0.05); a backend whose numbers were
26
+ * declared over the labels alone passes `support: 1`. Fails typed when
27
+ * none of the offered labels carries any mass, so a caller never branches
28
+ * on a silent uniform distribution.
29
+ */
30
+ export const normalizeLabelProbabilities = (
31
+ labels: ReadonlyArray<string>,
32
+ observed: Readonly<Record<string, number>>,
33
+ method: LabelMethod,
34
+ extra: {
35
+ readonly usage?: LabelDistribution["usage"]
36
+ readonly model?: string
37
+ readonly support?: number
38
+ } = {}
39
+ ): Effect.Effect<LabelDistribution, ParseError> => {
40
+ const kept = labels.map((label) => [label, Math.max(0, observed[label] ?? 0)] as const)
41
+ const total = kept.reduce((sum, [, value]) => sum + value, 0)
42
+ if (!(total > 0)) {
43
+ return Effect.fail(
44
+ ParseError.make({
45
+ message: `none of the offered labels was observed: ${labels.join(", ")}`,
46
+ raw: raw(JSON.stringify(observed))
47
+ })
48
+ )
49
+ }
50
+ return Effect.succeed(
51
+ LabelDistribution.make({
52
+ probabilities: Object.fromEntries(kept.map(([label, value]) => [label, value / total])),
53
+ method,
54
+ support: Math.max(0, Math.min(1, extra.support ?? total)),
55
+ ...(extra.usage === undefined ? {} : { usage: extra.usage }),
56
+ ...(extra.model === undefined ? {} : { model: extra.model })
57
+ })
58
+ )
59
+ }
60
+
61
+ export class VerbalizedLabels extends Schema.Class<VerbalizedLabels>("VerbalizedLabels")({
62
+ label: Schema.String,
63
+ probabilities: Schema.Record(Schema.String, Schema.Number)
64
+ }) {}
65
+
66
+ export const verbalizedLabelsJsonSchema = (labels: ReadonlyArray<string>): JsonSchema => ({
67
+ type: "object",
68
+ properties: {
69
+ label: { type: "string", enum: [...labels] },
70
+ probabilities: {
71
+ type: "object",
72
+ properties: Object.fromEntries(labels.map((label) => [label, { type: "number" }])),
73
+ required: [...labels],
74
+ additionalProperties: false
75
+ }
76
+ },
77
+ required: ["label", "probabilities"],
78
+ additionalProperties: false
79
+ })
80
+
81
+ export const verbalizedLabelsPrompt = (prompt: string, labels: ReadonlyArray<string>): string =>
82
+ `${prompt}\n\nReply with JSON only: {"label": <the one label you choose>, "probabilities": {<a probability for every label, summing to 1>}}. Labels: ${labels.join(", ")}.`
83
+
84
+ /**
85
+ * The default `scoreLabels` for a connector without log-probabilities: one
86
+ * schema-constrained JSON reply, renormalized over the offered labels.
87
+ */
88
+ export const verbalizedScoreLabels =
89
+ (executeStructuredWithUsage: LlmServiceShape["executeStructuredWithUsage"]) =>
90
+ (prompt: string, labels: ReadonlyArray<string>): Effect.Effect<LabelDistribution, LlmError> =>
91
+ executeStructuredWithUsage(
92
+ verbalizedLabelsPrompt(prompt, labels),
93
+ VerbalizedLabels,
94
+ verbalizedLabelsJsonSchema(labels)
95
+ ).pipe(
96
+ Effect.flatMap(([reply, usage, model]) =>
97
+ // A reply that names a label but gives it no mass contradicts itself;
98
+ // failing here is what lets the judgment layer retry or hold, instead
99
+ // of a manufactured 1.0 that could wave a review through.
100
+ labels.includes(reply.label) && !((reply.probabilities[reply.label] ?? 0) > 0)
101
+ ? Effect.fail(
102
+ ParseError.make({
103
+ message: `the model chose "${reply.label}" but gave it no probability`,
104
+ raw: raw(JSON.stringify(reply.probabilities))
105
+ })
106
+ )
107
+ : normalizeLabelProbabilities(labels, reply.probabilities, "verbalized", {
108
+ // Declared over the labels alone: support is by construction.
109
+ support: 1,
110
+ ...(usage === undefined ? {} : { usage }),
111
+ ...(model === undefined ? {} : { model })
112
+ })
113
+ )
114
+ )
115
+
116
+ /** For fakes and adapters that cannot classify: fails typed instead of guessing. */
117
+ export const unsupportedScoreLabels: LlmServiceShape["scoreLabels"] = (_prompt, _labels) =>
118
+ Effect.fail(InvalidRequestError.make({ message: "label scoring is not supported here" }))
119
+
120
+ export class VerbalizedLabelSequence extends Schema.Class<VerbalizedLabelSequence>(
121
+ "VerbalizedLabelSequence"
122
+ )({
123
+ answers: Schema.Array(VerbalizedLabels)
124
+ }) {}
125
+
126
+ export const verbalizedLabelSequenceJsonSchema = (
127
+ labelSets: ReadonlyArray<ReadonlyArray<string>>
128
+ ): JsonSchema => ({
129
+ type: "object",
130
+ properties: {
131
+ answers: {
132
+ type: "array",
133
+ minItems: labelSets.length,
134
+ maxItems: labelSets.length,
135
+ items: {
136
+ type: "object",
137
+ properties: {
138
+ label: { type: "string" },
139
+ probabilities: { type: "object", additionalProperties: { type: "number" } }
140
+ },
141
+ required: ["label", "probabilities"]
142
+ }
143
+ }
144
+ },
145
+ required: ["answers"],
146
+ additionalProperties: false
147
+ })
148
+
149
+ export const verbalizedLabelSequencePrompt = (
150
+ prompt: string,
151
+ labelSets: ReadonlyArray<ReadonlyArray<string>>
152
+ ): string =>
153
+ `${prompt}\n\nReply with JSON only: {"answers": [<one entry per question, in order>]}, each entry {"label": <the one label you choose>, "probabilities": {<a probability for every label of that question, summing to 1>}}. Labels per question: ${labelSets
154
+ .map((labels, index) => `${index + 1}: ${labels.join(", ")}`)
155
+ .join("; ")}.`
156
+
157
+ /**
158
+ * The derived `scoreLabelSequence`: one schema-constrained JSON reply with
159
+ * an answer per question. A missing position, or a chosen label with no
160
+ * mass, fails that position only.
161
+ */
162
+ export const verbalizedScoreLabelSequence =
163
+ (executeStructuredWithUsage: LlmServiceShape["executeStructuredWithUsage"]) =>
164
+ (
165
+ prompt: string,
166
+ labelSets: ReadonlyArray<ReadonlyArray<string>>
167
+ ): Effect.Effect<LabelSequence, LlmError> =>
168
+ executeStructuredWithUsage(
169
+ verbalizedLabelSequencePrompt(prompt, labelSets),
170
+ VerbalizedLabelSequence,
171
+ verbalizedLabelSequenceJsonSchema(labelSets)
172
+ ).pipe(
173
+ Effect.flatMap(([reply, usage, model]) =>
174
+ Effect.forEach(labelSets, (labels, index) => {
175
+ const answer = reply.answers[index]
176
+ const entry: Effect.Effect<LabelDistribution, ParseError> =
177
+ answer === undefined
178
+ ? Effect.fail(
179
+ ParseError.make({
180
+ message: `no answer for question ${index + 1} of ${labelSets.length}`,
181
+ raw: raw(JSON.stringify(reply.answers))
182
+ })
183
+ )
184
+ : labels.includes(answer.label) && !((answer.probabilities[answer.label] ?? 0) > 0)
185
+ ? Effect.fail(
186
+ ParseError.make({
187
+ message: `question ${index + 1}: the model chose "${answer.label}" but gave it no probability`,
188
+ raw: raw(JSON.stringify(answer.probabilities))
189
+ })
190
+ )
191
+ : normalizeLabelProbabilities(labels, answer.probabilities, "verbalized", {
192
+ support: 1
193
+ })
194
+ return Effect.result(entry)
195
+ }).pipe(
196
+ Effect.map(
197
+ (entries): LabelSequence => ({
198
+ entries,
199
+ ...(usage === undefined ? {} : { usage }),
200
+ ...(model === undefined ? {} : { model })
201
+ })
202
+ )
203
+ )
204
+ )
205
+ )
206
+
207
+ /** Convenience for callers that want the sequence's successes only. */
208
+ export const sequenceSuccesses = (
209
+ sequence: LabelSequence
210
+ ): ReadonlyArray<LabelDistribution | undefined> =>
211
+ sequence.entries.map((entry) => (Result.isSuccess(entry) ? entry.success : undefined))
package/src/LlmService.ts CHANGED
@@ -2,9 +2,11 @@ import * as Context from "effect/Context"
2
2
  import type * as Effect from "effect/Effect"
3
3
  import type * as Stream from "effect/Stream"
4
4
  import type * as Schema from "effect/Schema"
5
- import type { LlmError } from "./Errors.ts"
5
+ import type * as Result from "effect/Result"
6
+ import type { LlmError, ParseError } from "./Errors.ts"
6
7
  import type {
7
8
  JsonSchema,
9
+ LabelDistribution,
8
10
  LlmChunk,
9
11
  Message,
10
12
  TokenUsage,
@@ -37,9 +39,43 @@ export interface LlmServiceShape {
37
39
  schema: Schema.ConstraintCodec<A, E, RD, RE>,
38
40
  jsonSchema: JsonSchema
39
41
  ) => Effect.Effect<StructuredResult<A>, LlmError, RD>
42
+ /**
43
+ * One atomic classification: the probability of each offered label given
44
+ * the prompt. Connectors with token log-probabilities answer in one forward
45
+ * pass; the rest derive it from `executeStructured` (see `LabelScoring`).
46
+ * Fails with `ParseError` when no offered label could be observed.
47
+ */
48
+ readonly scoreLabels: (
49
+ prompt: string,
50
+ labels: ReadonlyArray<string>
51
+ ) => Effect.Effect<LabelDistribution, LlmError>
52
+ /**
53
+ * Several label questions answered in one call over a shared prompt
54
+ * prefix (the judgment layer's `shared-prefix` batching): the reply is one
55
+ * label per question in order, and each position is read as its own
56
+ * distribution. Optional, unlike `scoreLabels`: it is a cost optimization
57
+ * that only backends with token log-probabilities implement natively, and
58
+ * the judgment layer derives a verbalized version from structured output
59
+ * when it is absent, so no fake or decorator has to carry it.
60
+ */
61
+ readonly scoreLabelSequence?: (
62
+ prompt: string,
63
+ labelSets: ReadonlyArray<ReadonlyArray<string>>
64
+ ) => Effect.Effect<LabelSequence, LlmError>
40
65
  readonly isAvailable: Effect.Effect<boolean>
41
66
  }
42
67
 
68
+ /**
69
+ * The reply to `scoreLabelSequence`: one outcome per question in order (a
70
+ * position that could not be read fails on its own, so the caller can fall
71
+ * back for that question only), plus the call's usage and model once.
72
+ */
73
+ export interface LabelSequence {
74
+ readonly entries: ReadonlyArray<Result.Result<LabelDistribution, ParseError>>
75
+ readonly usage?: TokenUsage
76
+ readonly model?: string
77
+ }
78
+
43
79
  export class LlmService extends Context.Service<LlmService, LlmServiceShape>()(
44
80
  "@llm4ts/core/LlmService"
45
81
  ) {}
package/src/Models.ts CHANGED
@@ -9,6 +9,7 @@ export const LlmProvider = Schema.Literals([
9
9
  "Anthropic",
10
10
  "LmStudio",
11
11
  "Ollama",
12
+ "MlxLm",
12
13
  "OpenCode",
13
14
  "Mock"
14
15
  ])
@@ -24,6 +25,7 @@ export const ConnectorIds = Object.freeze({
24
25
  GeminiApi: new ConnectorId({ value: "gemini-api" }),
25
26
  LmStudio: new ConnectorId({ value: "lm-studio" }),
26
27
  Ollama: new ConnectorId({ value: "ollama" }),
28
+ MlxLm: new ConnectorId({ value: "mlx-lm" }),
27
29
  ClaudeCli: new ConnectorId({ value: "claude-cli" }),
28
30
  GeminiCli: new ConnectorId({ value: "gemini-cli" }),
29
31
  OpenCode: new ConnectorId({ value: "opencode" }),
@@ -41,7 +43,8 @@ export const apiConnectorIds: ReadonlyArray<ConnectorId> = Object.freeze([
41
43
  ConnectorIds.Anthropic,
42
44
  ConnectorIds.GeminiApi,
43
45
  ConnectorIds.LmStudio,
44
- ConnectorIds.Ollama
46
+ ConnectorIds.Ollama,
47
+ ConnectorIds.MlxLm
45
48
  ])
46
49
 
47
50
  export const cliConnectorIds: ReadonlyArray<ConnectorId> = Object.freeze([
@@ -77,6 +80,8 @@ export const defaultBaseUrl = (provider: LlmProvider): string | undefined => {
77
80
  return "http://localhost:1234/v1"
78
81
  case "Ollama":
79
82
  return "http://localhost:11434"
83
+ case "MlxLm":
84
+ return "http://localhost:8080"
80
85
  case "OpenCode":
81
86
  return "http://localhost:4096"
82
87
  }
@@ -88,6 +93,7 @@ const connectorProviderTable: Readonly<Record<string, LlmProvider>> = {
88
93
  "gemini-api": "GeminiApi",
89
94
  "lm-studio": "LmStudio",
90
95
  ollama: "Ollama",
96
+ "mlx-lm": "MlxLm",
91
97
  opencode: "OpenCode",
92
98
  "gemini-cli": "GeminiCli",
93
99
  mock: "Mock"
@@ -115,6 +121,8 @@ export const providerConnectorId = (provider: LlmProvider): ConnectorId => {
115
121
  return ConnectorIds.LmStudio
116
122
  case "Ollama":
117
123
  return ConnectorIds.Ollama
124
+ case "MlxLm":
125
+ return ConnectorIds.MlxLm
118
126
  case "OpenCode":
119
127
  return ConnectorIds.OpenCode
120
128
  case "Mock":
@@ -255,6 +263,37 @@ export type InteractionSupport = typeof InteractionSupport.Type
255
263
  export const ReadOnlyEnforcement = Schema.Literals(["enforced", "advisory", "ignored"])
256
264
  export type ReadOnlyEnforcement = typeof ReadOnlyEnforcement.Type
257
265
 
266
+ /**
267
+ * How a connector produces the per-label probabilities behind `scoreLabels`:
268
+ * - "logprobs": read off the backend's token log-probabilities in one
269
+ * forward pass (a real distribution, still uncalibrated).
270
+ * - "verbalized": the model writes the numbers itself in a JSON reply
271
+ * (typically overconfident; hold to a higher bar).
272
+ * - "none": the connector cannot answer label questions at all.
273
+ */
274
+ export const LabelProbabilities = Schema.Literals(["logprobs", "verbalized", "none"])
275
+ export type LabelProbabilities = typeof LabelProbabilities.Type
276
+
277
+ /** How a label distribution's numbers were extracted (see `LabelProbabilities`). */
278
+ export const LabelMethod = Schema.Literals(["logprobs", "verbalized", "sampled"])
279
+ export type LabelMethod = typeof LabelMethod.Type
280
+
281
+ /**
282
+ * A probability per label, normalized over the labels the caller offered.
283
+ * `support` is the probability mass the backend actually placed on those
284
+ * labels before renormalization (1 when the numbers were declared over the
285
+ * labels alone): a distribution renormalized from a sliver of mass is not
286
+ * a confident one, and callers must not treat it as one.
287
+ * `usage` is whatever the backend reported for the call, if anything.
288
+ */
289
+ export class LabelDistribution extends Schema.Class<LabelDistribution>("LabelDistribution")({
290
+ probabilities: Schema.Record(Schema.String, Schema.Number),
291
+ method: LabelMethod,
292
+ support: Schema.Number,
293
+ usage: Schema.optionalKey(TokenUsage),
294
+ model: Schema.optionalKey(Schema.String)
295
+ }) {}
296
+
258
297
  export class ConnectorCapabilities extends Schema.Class<ConnectorCapabilities>(
259
298
  "ConnectorCapabilities"
260
299
  )({
@@ -265,6 +304,9 @@ export class ConnectorCapabilities extends Schema.Class<ConnectorCapabilities>(
265
304
  approval: Schema.Boolean.pipe(Schema.withConstructorDefault(Effect.succeed(false))),
266
305
  structuredOutput: Schema.Boolean.pipe(Schema.withConstructorDefault(Effect.succeed(true))),
267
306
  usageReporting: Schema.Boolean.pipe(Schema.withConstructorDefault(Effect.succeed(true))),
307
+ labelProbabilities: LabelProbabilities.pipe(
308
+ Schema.withConstructorDefault(Effect.succeed<LabelProbabilities>("verbalized"))
309
+ ),
268
310
  readOnlyEnforcement: ReadOnlyEnforcement.pipe(
269
311
  Schema.withConstructorDefault(Effect.succeed<ReadOnlyEnforcement>("advisory"))
270
312
  )
package/src/eval/Judge.ts CHANGED
@@ -1,5 +1,8 @@
1
1
  import * as Effect from "effect/Effect"
2
2
  import * as Schema from "effect/Schema"
3
+ import { ProviderError } from "../Errors.ts"
4
+ import type { JudgmentShape } from "../judgment/Judgment.ts"
5
+ import { score, type ScoreQuestion, type State } from "../judgment/Schemas.ts"
3
6
  import type { LlmServiceShape } from "../LlmService.ts"
4
7
  import type { JsonSchema } from "../Models.ts"
5
8
  import { DimensionScore, EvalResult } from "./Eval.ts"
@@ -90,3 +93,74 @@ export const judge = (
90
93
  .executeStructured(buildPrompt(system, dimensions, sample), JudgeResponse, judgeJsonSchema)
91
94
  .pipe(Effect.map((response) => toResult(dimensions, response)))
92
95
  )
96
+
97
+ /**
98
+ * A dimension as a Score question: one level per rubric point, so the
99
+ * judgment returns a distribution over the scale instead of one integer.
100
+ */
101
+ export const dimensionQuestion = (dimension: Dimension): ScoreQuestion =>
102
+ score(
103
+ `${dimension.name}: ${dimension.rubric}`,
104
+ Array.from({ length: dimension.maxScore + 1 }, (_, level) => ({
105
+ level,
106
+ of: dimension.maxScore,
107
+ meaning:
108
+ level === 0
109
+ ? "does not meet the rubric at all"
110
+ : level === dimension.maxScore
111
+ ? "fully meets the rubric"
112
+ : `partially meets the rubric (${level} of ${dimension.maxScore})`
113
+ }))
114
+ )
115
+
116
+ export const sampleState = (sample: Sample): State => ({
117
+ ...(sample.query === undefined ? {} : { query: sample.query }),
118
+ ...(sample.context === undefined ? {} : { context: sample.context }),
119
+ response: sample.response,
120
+ ...(sample.expected === undefined ? {} : { expected: sample.expected })
121
+ })
122
+
123
+ /**
124
+ * The rubric judge over the Judgment service (ADR 0017, consumer 1): every
125
+ * dimension is an independent Score question over the same sample. A
126
+ * dimension the backend could not answer scores 0 with the reason recorded,
127
+ * matching the generative judge's treatment of a missing score.
128
+ */
129
+ export const judgeWithJudgment = (
130
+ judgment: JudgmentShape,
131
+ dimensions: ReadonlyArray<Dimension>
132
+ ): Evaluator<Sample> =>
133
+ makeEvaluator((sample) =>
134
+ judgment
135
+ .judge({
136
+ state: sampleState(sample),
137
+ questions: Object.fromEntries(
138
+ dimensions.map((dimension) => [dimension.name, dimensionQuestion(dimension)])
139
+ )
140
+ })
141
+ .pipe(
142
+ Effect.mapError((error) =>
143
+ ProviderError.make({ message: `judgment backend failed: ${error.message}`, cause: error })
144
+ ),
145
+ Effect.map((result) =>
146
+ EvalResult.make({
147
+ scores: dimensions.map((dimension) => {
148
+ const answer = result.answers[dimension.name]
149
+ if (answer === undefined || answer.type !== "score") {
150
+ const failure = result.failures.find((entry) => entry.key === dimension.name)
151
+ return DimensionScore.make({
152
+ name: dimension.name,
153
+ score: 0,
154
+ reasoning: failure === undefined ? "missing" : `failed: ${failure.reason}`
155
+ })
156
+ }
157
+ return DimensionScore.make({
158
+ name: dimension.name,
159
+ score: Math.max(0, Math.min(dimension.maxScore, Math.round(answer.score))),
160
+ reasoning: `${answer.origin.method} judgment (${answer.origin.backend}), confidence ${answer.confidence.toFixed(2)}, support ${answer.support.toFixed(2)}, expected level ${answer.score.toFixed(2)}`
161
+ })
162
+ })
163
+ })
164
+ )
165
+ )
166
+ )
@@ -0,0 +1,90 @@
1
+ import * as Effect from "effect/Effect"
2
+ import * as Layer from "effect/Layer"
3
+ import * as Ref from "effect/Ref"
4
+ import { Judgment, type JudgmentInput, type JudgmentShape } from "./Judgment.ts"
5
+ import {
6
+ choiceAnswer,
7
+ JudgmentRequest,
8
+ JudgmentResult,
9
+ origins,
10
+ QuestionFailure,
11
+ scoreAnswer,
12
+ truthAnswer,
13
+ type Answer,
14
+ type Question
15
+ } from "./Schemas.ts"
16
+
17
+ /**
18
+ * `FakeJudgment`: deterministic answers for tests. A plan maps question keys
19
+ * to canned answers (or a failure reason); unplanned questions get a
20
+ * predictable default so a consumer test can run without listing every key.
21
+ */
22
+
23
+ export interface FakeJudgmentPlan {
24
+ readonly answers?: Readonly<Record<string, Answer>>
25
+ readonly failures?: Readonly<Record<string, string>>
26
+ /** Truth probability for unplanned truth questions (default 1). */
27
+ readonly defaultTruth?: number
28
+ }
29
+
30
+ export interface FakeJudgment {
31
+ readonly judgment: JudgmentShape
32
+ readonly recorded: Effect.Effect<ReadonlyArray<JudgmentRequest>>
33
+ }
34
+
35
+ const defaultAnswer = (question: Question, defaultTruth: number): Answer => {
36
+ switch (question.type) {
37
+ case "choice": {
38
+ const keys = Object.keys(question.criteria)
39
+ return choiceAnswer(
40
+ Object.fromEntries(keys.map((key, index) => [key, index === 0 ? 1 : 0])),
41
+ origins.fake()
42
+ )
43
+ }
44
+ case "score":
45
+ return scoreAnswer(
46
+ question,
47
+ Object.fromEntries(
48
+ question.criteria.map((_, index) => [String(index), index === 0 ? 1 : 0])
49
+ ),
50
+ origins.fake()
51
+ )
52
+ case "truth":
53
+ return truthAnswer(defaultTruth, origins.fake())
54
+ }
55
+ }
56
+
57
+ export const makeFakeJudgment = Effect.fn("@llm4ts/core/judgment/FakeJudgment.make")(function* (
58
+ plan: FakeJudgmentPlan = {}
59
+ ): Effect.fn.Return<FakeJudgment> {
60
+ const requests = yield* Ref.make<ReadonlyArray<JudgmentRequest>>([])
61
+ const judge = (input: JudgmentInput): Effect.Effect<JudgmentResult> =>
62
+ Ref.update(requests, (all) => [
63
+ ...all,
64
+ JudgmentRequest.make({ state: input.state, questions: input.questions })
65
+ ]).pipe(
66
+ Effect.map(() => {
67
+ const answers: Record<string, Answer> = {}
68
+ const failures: Array<QuestionFailure> = []
69
+ for (const [key, question] of Object.entries(input.questions)) {
70
+ const failure = plan.failures?.[key]
71
+ if (failure !== undefined) {
72
+ failures.push(QuestionFailure.make({ key, reason: failure }))
73
+ continue
74
+ }
75
+ answers[key] = plan.answers?.[key] ?? defaultAnswer(question, plan.defaultTruth ?? 1)
76
+ }
77
+ return JudgmentResult.make({ answers, failures, backend: "fake" })
78
+ })
79
+ )
80
+ return {
81
+ judgment: { backend: "fake", identity: "fake", judge },
82
+ recorded: Ref.get(requests)
83
+ }
84
+ })
85
+
86
+ export const FakeJudgmentLive = (plan?: FakeJudgmentPlan): Layer.Layer<Judgment> =>
87
+ Layer.effect(
88
+ Judgment,
89
+ Effect.map(makeFakeJudgment(plan), (fake) => fake.judgment)
90
+ )
@@ -0,0 +1,123 @@
1
+ import * as Context from "effect/Context"
2
+ import * as Effect from "effect/Effect"
3
+ import * as Schema from "effect/Schema"
4
+ import {
5
+ ChoiceAnswer,
6
+ ScoreAnswer,
7
+ TruthAnswer,
8
+ type Answer,
9
+ type JudgmentBackend,
10
+ type JudgmentResult,
11
+ type Question,
12
+ type State
13
+ } from "./Schemas.ts"
14
+
15
+ /**
16
+ * The Judgment service (ADR 0017): evaluate typed questions against one
17
+ * State and return typed answers with probabilities and their origin. It
18
+ * answers; it never decides. Policy (thresholds, escalation, caching) lives
19
+ * in the flow package.
20
+ */
21
+
22
+ export class JudgmentBackendError extends Schema.TaggedError<JudgmentBackendError>()(
23
+ "JudgmentBackendError",
24
+ {
25
+ backend: Schema.String,
26
+ message: Schema.String,
27
+ cause: Schema.optionalKey(Schema.Defect())
28
+ }
29
+ ) {}
30
+
31
+ /** Raised by the typed accessors when a key is missing or of another kind. */
32
+ export class AnswerMismatch extends Schema.TaggedError<AnswerMismatch>()("AnswerMismatch", {
33
+ key: Schema.String,
34
+ expected: Schema.String,
35
+ actual: Schema.optionalKey(Schema.String),
36
+ reason: Schema.optionalKey(Schema.String)
37
+ }) {
38
+ get message(): string {
39
+ return this.actual === undefined
40
+ ? `no ${this.expected} answer for "${this.key}"${this.reason === undefined ? "" : `: ${this.reason}`}`
41
+ : `answer "${this.key}" is a ${this.actual}, not a ${this.expected}`
42
+ }
43
+ }
44
+
45
+ export const JudgmentError = Schema.Union([JudgmentBackendError, AnswerMismatch])
46
+ export type JudgmentError = typeof JudgmentError.Type
47
+
48
+ export interface JudgmentInput {
49
+ readonly state: State
50
+ readonly questions: Readonly<Record<string, Question>>
51
+ }
52
+
53
+ export interface JudgmentShape {
54
+ readonly backend: JudgmentBackend
55
+ /**
56
+ * Backend plus checkpoint, e.g. `llm:mlx-lm:/models/qwen3-4b` or
57
+ * `typesafe:jev-1.13.0`: what a cached answer is keyed on, so swapping a
58
+ * local model never reuses its predecessor's answers.
59
+ */
60
+ readonly identity: string
61
+ /**
62
+ * Answer every question independently over the same state. A question the
63
+ * backend could not answer lands in `failures`; the request as a whole
64
+ * fails only when the backend itself is unreachable or rejects it.
65
+ */
66
+ readonly judge: (input: JudgmentInput) => Effect.Effect<JudgmentResult, JudgmentBackendError>
67
+ }
68
+
69
+ export class Judgment extends Context.Service<Judgment, JudgmentShape>()(
70
+ "@llm4ts/core/judgment/Judgment"
71
+ ) {}
72
+
73
+ const answerOf = (
74
+ result: JudgmentResult,
75
+ key: string,
76
+ expected: Answer["type"]
77
+ ): Effect.Effect<Answer, AnswerMismatch> => {
78
+ const answer = result.answers[key]
79
+ if (answer === undefined) {
80
+ const failure = result.failures.find((entry) => entry.key === key)
81
+ return Effect.fail(
82
+ AnswerMismatch.make({
83
+ key,
84
+ expected,
85
+ ...(failure === undefined ? {} : { reason: failure.reason })
86
+ })
87
+ )
88
+ }
89
+ return answer.type === expected
90
+ ? Effect.succeed(answer)
91
+ : Effect.fail(AnswerMismatch.make({ key, expected, actual: answer.type }))
92
+ }
93
+
94
+ /** Typed accessors: the honest way to read an answer by key. */
95
+ export const choiceOf = (
96
+ result: JudgmentResult,
97
+ key: string
98
+ ): Effect.Effect<ChoiceAnswer, AnswerMismatch> =>
99
+ Effect.flatMap(answerOf(result, key, "choice"), (answer) =>
100
+ answer instanceof ChoiceAnswer
101
+ ? Effect.succeed(answer)
102
+ : Effect.fail(AnswerMismatch.make({ key, expected: "choice", actual: answer.type }))
103
+ )
104
+
105
+ export const scoreOf = (
106
+ result: JudgmentResult,
107
+ key: string
108
+ ): Effect.Effect<ScoreAnswer, AnswerMismatch> =>
109
+ Effect.flatMap(answerOf(result, key, "score"), (answer) =>
110
+ answer instanceof ScoreAnswer
111
+ ? Effect.succeed(answer)
112
+ : Effect.fail(AnswerMismatch.make({ key, expected: "score", actual: answer.type }))
113
+ )
114
+
115
+ export const truthOf = (
116
+ result: JudgmentResult,
117
+ key: string
118
+ ): Effect.Effect<TruthAnswer, AnswerMismatch> =>
119
+ Effect.flatMap(answerOf(result, key, "truth"), (answer) =>
120
+ answer instanceof TruthAnswer
121
+ ? Effect.succeed(answer)
122
+ : Effect.fail(AnswerMismatch.make({ key, expected: "truth", actual: answer.type }))
123
+ )