mohdel 3.7.0 → 3.8.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -472,6 +472,7 @@ QWEN_API_SK=sk-...
472
472
  XIAOMI_API_SK=...
473
473
  META_API_SK=...
474
474
  COHERE_API_SK=...
475
+ TYPESAFE_API_SK=...
475
476
  MOHDEL_LOCAL_API_SK=...
476
477
  ```
477
478
 
@@ -515,6 +516,7 @@ What each provider supports through mohdel's unified interface:
515
516
  | OpenRouter | Yes | Yes | Yes | No | Varies | Meta-provider; `providerOptions.openrouter` for routing prefs |
516
517
  | Local | Yes | Yes | Yes | No | No | Any OpenAI-compatible server; endpoint is the catalog entry's `baseURL` |
517
518
  | Cohere | n/a | n/a | n/a | n/a | n/a | Embeddings only: no chat models reach mohdel through it |
519
+ | TypeSafe | n/a | n/a | n/a | n/a | n/a | Evaluation only (`evaluate()`): typed questions answered with probabilities |
518
520
  | Novita | Yes | Yes | Yes | No | Yes (`reasoning_content`) | Prices in the model list; text via the shared chat-completions path, separate image adapter |
519
521
 
520
522
  Adapter capability ≠ model capability — whether a given model accepts images, tools, or thinking effort depends on the model spec in `curated.json`. The adapter passes through what the envelope supplies; the provider rejects unsupported combos.
@@ -87,10 +87,11 @@
87
87
  "enum": [
88
88
  "model",
89
89
  "image",
90
- "transcription"
90
+ "transcription",
91
+ "evaluation"
91
92
  ],
92
93
  "default": "model",
93
- "description": "'model' for chat/completion, 'image' for image generation, 'transcription' for speech-to-text."
94
+ "description": "'model' for chat/completion, 'image' for image generation, 'transcription' for speech-to-text, 'evaluation' for typed-question answering (evaluate())."
94
95
  },
95
96
  "label": {
96
97
  "type": "string",
@@ -550,6 +551,11 @@
550
551
  "type": "string",
551
552
  "description": "Symbolic role sent when the caller names none. Required for providers that make the parameter mandatory (Cohere v3+)."
552
553
  },
554
+ "evaluationTypes": {
555
+ "type": "array",
556
+ "items": { "type": "string", "enum": ["binary", "choice", "score"] },
557
+ "description": "Question types an evaluation model answers. A question of another type fails before dispatch. Without it, every type is sent and the provider decides."
558
+ },
553
559
  "rpmLimit": {
554
560
  "type": "integer",
555
561
  "minimum": 1,
@@ -0,0 +1,47 @@
1
+ /**
2
+ * Send an EvaluateEnvelope to thin-gate's `POST /v1/evaluate`. One-shot: a
3
+ * single JSON response body.
4
+ *
5
+ * @module client/call_evaluation
6
+ */
7
+
8
+ import { requestUnix } from './transport.js'
9
+ import { readAll, parseErrorBody } from './response.js'
10
+ import { MohdelError } from '#core'
11
+
12
+ /**
13
+ * @param {import('#core/evaluation.js').EvaluateEnvelope} envelope
14
+ * @param {object} options
15
+ * @param {string} options.socketPath
16
+ * @param {AbortSignal} [options.signal]
17
+ * @param {string} [options.path] HTTP path; defaults to '/v1/evaluate'
18
+ * @param {Record<string, string>} [options.headers] sent with the request, for a router in front of the gate;
19
+ * `content-type`, `content-length`, `transfer-encoding`, `connection` and `host` are the transport's
20
+ * @returns {Promise<import('#core/evaluation.js').EvaluateResult>}
21
+ */
22
+ export async function callEvaluation (envelope, { socketPath, signal, path = '/v1/evaluate', headers }) {
23
+ const res = await requestUnix({
24
+ socketPath,
25
+ path,
26
+ method: 'POST',
27
+ body: envelope,
28
+ signal,
29
+ headers
30
+ })
31
+
32
+ const body = await readAll(res)
33
+
34
+ if (res.statusCode !== 200) {
35
+ throw MohdelError.fromJSON(parseErrorBody(body, res.statusCode ?? 0))
36
+ }
37
+
38
+ try {
39
+ return JSON.parse(body)
40
+ } catch (e) {
41
+ throw new MohdelError(`gate returned an unparseable evaluate body: ${body.slice(0, 200)}`, {
42
+ type: 'PROTOCOL_INVALID_RESPONSE',
43
+ severity: 'error',
44
+ retryable: false
45
+ })
46
+ }
47
+ }
@@ -18,4 +18,5 @@ export { coalesce } from './coalesce.js'
18
18
  export { callImage } from './call_image.js'
19
19
  export { callTranscription } from './call_transcription.js'
20
20
  export { callEmbedding } from './call_embedding.js'
21
+ export { callEvaluation } from './call_evaluation.js'
21
22
  export { resolveGateBinary } from './gate-binary.js'
@@ -0,0 +1,105 @@
1
+ /**
2
+ * Evaluation envelope and result: one `state`, a map of typed questions, one
3
+ * typed answer per question.
4
+ *
5
+ * @module core/evaluation
6
+ */
7
+
8
+ /**
9
+ * @typedef {object} EvaluateEnvelope
10
+ * @property {string} callId
11
+ * @property {string} authId
12
+ * @property {import('./envelope.js').Auth} auth
13
+ * @property {string} [traceparent]
14
+ * @property {string} [baggage]
15
+ * @property {string} model
16
+ * @property {string | object | any[]} state
17
+ * @property {Record<string, Question>} questions
18
+ * Answers come back under the same ids.
19
+ */
20
+
21
+ /**
22
+ * @typedef {BinaryQuestion | ChoiceQuestion | ScoreQuestion} Question
23
+ *
24
+ * `instructions` and every criterion description may be a string, an object
25
+ * or an array.
26
+ */
27
+
28
+ /**
29
+ * @typedef {object} BinaryQuestion
30
+ * @property {'binary'} type
31
+ * @property {any} instructions
32
+ * @property {{yes: any, no: any}} [criteria]
33
+ */
34
+
35
+ /**
36
+ * @typedef {object} ChoiceQuestion
37
+ * @property {'choice'} type
38
+ * @property {any} instructions
39
+ * @property {Record<string, any>} criteria
40
+ * Option name to its description, or null. At least two options.
41
+ */
42
+
43
+ /**
44
+ * @typedef {object} ScoreQuestion
45
+ * @property {'score'} type
46
+ * @property {any} instructions
47
+ * @property {any[]} criteria
48
+ * Level descriptions, lowest first; level `i` is worth `i`. At least two.
49
+ */
50
+
51
+ /**
52
+ * @typedef {BinaryAnswer | ChoiceAnswer | ScoreAnswer} Answer
53
+ */
54
+
55
+ /**
56
+ * @typedef {object} BinaryAnswer
57
+ * @property {'binary'} type
58
+ * @property {number} probability
59
+ * Of yes.
60
+ */
61
+
62
+ /**
63
+ * @typedef {object} ChoiceAnswer
64
+ * @property {'choice'} type
65
+ * @property {string} choice
66
+ * @property {Record<string, number>} probabilities
67
+ * @property {number} [confidence]
68
+ * Absent when the provider reports none.
69
+ */
70
+
71
+ /**
72
+ * @typedef {object} ScoreAnswer
73
+ * @property {'score'} type
74
+ * @property {number} score
75
+ * Probability-weighted level; can fall between two.
76
+ * @property {number[]} probabilities
77
+ * One per level, in `criteria` order.
78
+ * @property {number} [confidence]
79
+ */
80
+
81
+ /**
82
+ * @typedef {object} EvaluateResult
83
+ * @property {'completed'} status
84
+ * @property {Record<string, Answer>} answers
85
+ * @property {string} upstreamModel
86
+ * The provider's id for the version that answered, which differs from the
87
+ * entry's `model` when that is an alias.
88
+ * @property {number} inputTokens
89
+ * @property {number} outputTokens
90
+ * @property {number} cost
91
+ * @property {{start: string, first: string, end: string}} timestamps
92
+ */
93
+
94
+ export const EVALUATE_ENVELOPE_FIELDS = Object.freeze([
95
+ 'callId',
96
+ 'authId',
97
+ 'auth',
98
+ 'traceparent',
99
+ 'baggage',
100
+ 'model',
101
+ 'state',
102
+ 'questions'
103
+ ])
104
+
105
+ export const QUESTION_TYPES = Object.freeze(['binary', 'choice', 'score'])
@@ -24,6 +24,7 @@ import { run } from '../session/run.js'
24
24
  import { runImage } from '../session/run_image.js'
25
25
  import { runTranscription } from '../session/run_transcription.js'
26
26
  import { runEmbedding } from '../session/run_embedding.js'
27
+ import { runEvaluation } from '../session/run_evaluation.js'
27
28
  import { markTrustedMedia } from '../session/adapters/_media.js'
28
29
  import { MohdelError, validateIds } from '#core'
29
30
  import { createRealtimeDeltaBuffer } from '../../src/lib/utils.js'
@@ -224,6 +225,44 @@ export async function runAnswerEmbedding ({ provider, model, modelKey, configura
224
225
  return out.result
225
226
  }
226
227
 
228
+ /**
229
+ * Run an `evaluate()` call through the /session runtime.
230
+ *
231
+ * @param {object} args
232
+ * @param {string} args.provider
233
+ * @param {string} args.model
234
+ * @param {string} [args.modelKey]
235
+ * @param {any} args.configuration
236
+ * @param {import('#core/evaluation.js').EvaluateEnvelope['state']} args.state
237
+ * @param {import('#core/evaluation.js').EvaluateEnvelope['questions']} args.questions
238
+ * @param {any} [args.options] `callId` / `authId` only.
239
+ * @param {any} [args.spec]
240
+ * @param {BridgeDeps} [deps]
241
+ * @returns {Promise<import('#core/evaluation.js').EvaluateResult>}
242
+ */
243
+ export async function runAnswerEvaluation ({ provider, model, modelKey, configuration, state, questions, options = {}, spec }, deps = {}) {
244
+ const callId = options.callId || newCallId()
245
+ const authId = options.authId || 'local'
246
+ assertValidIds(callId, authId, `${provider}/${model}`)
247
+
248
+ const envelope = {
249
+ callId,
250
+ authId,
251
+ auth: configToAuth(configuration),
252
+ model: `${provider}/${model}`,
253
+ state,
254
+ questions
255
+ }
256
+
257
+ const out = await runEvaluation(envelope, {
258
+ ...deps,
259
+ ...(modelKey ? { modelKey } : {}),
260
+ ...(spec ? { spec } : {})
261
+ })
262
+ if (!out.ok) throw MohdelError.fromJSON(out.error, { provider, model })
263
+ return out.result
264
+ }
265
+
227
266
  /**
228
267
  * @param {object} args
229
268
  * @param {string} args.modelKey Mohdel catalog key `<provider>/<bare>`. The
@@ -262,6 +262,11 @@ async function * runStreaming (envelope, client, args, config, start, deps) {
262
262
  return
263
263
  }
264
264
 
265
+ if (finishReason === null) {
266
+ yield { type: 'error', error: classifyProviderError(new Error(`${config.provider} stream ended before completion`), envelope.auth?.key, { provider: config.provider }) }
267
+ return
268
+ }
269
+
265
270
  const collectedToolCalls = Object.values(toolCallAccum).map(tc => ({
266
271
  id: tc.id,
267
272
  function: { name: tc.name, arguments: tc.arguments }
@@ -469,9 +469,11 @@ export function typedError (message, type, retryable, detail) {
469
469
  * @param {number} status
470
470
  * @param {string} message
471
471
  * @param {string} [detail]
472
+ * @param {string} [key] masked out of `detail`
472
473
  * @returns {Error & {typed: import('#core/errors.js').TypedError}}
473
474
  */
474
- export function fromHttpStatus (status, message, detail) {
475
+ export function fromHttpStatus (status, message, detail, key) {
475
476
  const typed = classifyProviderError({ status })
476
- return typedError(typed.message, typed.type, typed.retryable, detail ? `${message}: ${detail}` : message)
477
+ const scrubbed = scrubKey(detail, key)
478
+ return typedError(typed.message, typed.type, typed.retryable, scrubbed ? `${message}: ${scrubbed}` : message)
477
479
  }
@@ -146,6 +146,16 @@ export function embeddingCostFor (envelope, spec, usage) {
146
146
  return unmetered(envelope.model) ? 0 : computeEmbeddingCost(spec, usage)
147
147
  }
148
148
 
149
+ /**
150
+ * @param {{model: string}} envelope
151
+ * @param {any} spec
152
+ * @param {{inputTokens: number, outputTokens: number}} usage
153
+ * @returns {number}
154
+ */
155
+ export function evaluationCostFor (envelope, spec, usage) {
156
+ return unmetered(envelope.model) ? 0 : computeCost(spec, usage)
157
+ }
158
+
149
159
  /**
150
160
  * Cost of a transcription call.
151
161
  *
@@ -61,7 +61,7 @@ export async function cohereEmbedding (envelope, deps = {}) {
61
61
 
62
62
  if (!res.ok) {
63
63
  const detail = await res.text().catch(() => '')
64
- throw fromHttpStatus(res.status, detail, envelope.auth?.key)
64
+ throw fromHttpStatus(res.status, 'embedding request failed', detail.slice(0, 500), envelope.auth?.key)
65
65
  }
66
66
 
67
67
  const payload = await res.json()
@@ -59,7 +59,7 @@ export async function geminiEmbedding (envelope, deps = {}) {
59
59
 
60
60
  if (!res.ok) {
61
61
  const detail = await res.text().catch(() => '')
62
- throw fromHttpStatus(res.status, detail, envelope.auth?.key)
62
+ throw fromHttpStatus(res.status, 'embedding request failed', detail.slice(0, 500), envelope.auth?.key)
63
63
  }
64
64
 
65
65
  const payload = await res.json()
@@ -56,7 +56,7 @@ export function createEmbeddingAdapter ({ baseURL, dimensionsField = 'dimensions
56
56
 
57
57
  if (!res.ok) {
58
58
  const detail = await res.text().catch(() => '')
59
- throw fromHttpStatus(res.status, detail, envelope.auth?.key)
59
+ throw fromHttpStatus(res.status, 'embedding request failed', detail.slice(0, 500), envelope.auth?.key)
60
60
  }
61
61
 
62
62
  const payload = await res.json()
@@ -0,0 +1,73 @@
1
+ /**
2
+ * Pre-dispatch checks every evaluation adapter runs. They reject what the
3
+ * gate's serde types reject, so a malformed envelope fails the same way
4
+ * in-process.
5
+ *
6
+ * @module session/adapters/evaluation/shared
7
+ */
8
+
9
+ import { typedError } from '../_errors.js'
10
+ import { QUESTION_TYPES } from '#core/evaluation.js'
11
+
12
+ const QUESTION_KEYS = new Set(['type', 'instructions', 'criteria'])
13
+ const BINARY_CRITERIA_KEYS = ['yes', 'no']
14
+
15
+ /**
16
+ * @param {import('#core/evaluation.js').EvaluateEnvelope} envelope
17
+ * @param {any} spec
18
+ */
19
+ export function checkEvaluation (envelope, spec) {
20
+ const { state, questions } = envelope
21
+ if (typeof state !== 'string' && !isObject(state) && !Array.isArray(state)) {
22
+ throw invalid('state must be a string, an object or an array')
23
+ }
24
+ if (!isObject(questions) || Object.keys(questions).length === 0) {
25
+ throw invalid('questions must be a non-empty object')
26
+ }
27
+ for (const [id, q] of Object.entries(questions)) {
28
+ checkQuestion(id, q, spec)
29
+ }
30
+ }
31
+
32
+ function checkQuestion (id, q, spec) {
33
+ if (!isObject(q)) throw invalid(`question '${id}' must be an object`)
34
+ const extra = Object.keys(q).filter(k => !QUESTION_KEYS.has(k))
35
+ if (extra.length) throw invalid(`question '${id}' has unknown field '${extra.join("', '")}'`)
36
+ if (!QUESTION_TYPES.includes(q.type)) {
37
+ throw invalid(`question '${id}' type must be one of ${QUESTION_TYPES.join(', ')}`)
38
+ }
39
+ if (Array.isArray(spec?.evaluationTypes) && !spec.evaluationTypes.includes(q.type)) {
40
+ throw typedError(
41
+ `question '${id}' is '${q.type}', which this model does not answer; it answers ${spec.evaluationTypes.join(', ')}`,
42
+ 'EVALUATE_QUESTION_TYPE_UNSUPPORTED',
43
+ false
44
+ )
45
+ }
46
+ if (q.instructions === undefined || q.instructions === null) {
47
+ throw invalid(`question '${id}' requires instructions`)
48
+ }
49
+
50
+ const c = q.criteria
51
+ if (q.type === 'binary') {
52
+ if (c === undefined) return
53
+ if (!isObject(c) || Object.keys(c).length !== 2 || !BINARY_CRITERIA_KEYS.every(k => c[k] !== undefined && c[k] !== null)) {
54
+ throw invalid(`question '${id}' criteria must have exactly 'yes' and 'no'`)
55
+ }
56
+ } else if (q.type === 'choice') {
57
+ if (!isObject(c) || Object.keys(c).length < 2) {
58
+ throw invalid(`question '${id}' criteria must name at least two options`)
59
+ }
60
+ } else if (!Array.isArray(c) || c.length < 2) {
61
+ throw invalid(`question '${id}' criteria must list at least two levels`)
62
+ }
63
+ }
64
+
65
+ /** @param {unknown} v */
66
+ function isObject (v) {
67
+ return typeof v === 'object' && v !== null && !Array.isArray(v)
68
+ }
69
+
70
+ /** @param {string} message */
71
+ function invalid (message) {
72
+ return typedError(message, 'EVALUATE_INPUT_INVALID', false)
73
+ }
@@ -0,0 +1,22 @@
1
+ /**
2
+ * Evaluation-adapter registry.
3
+ *
4
+ * @module session/adapters/evaluation
5
+ */
6
+
7
+ import { typesafeEvaluation } from './typesafe.js'
8
+
9
+ const EVALUATION_ADAPTERS = {
10
+ typesafe: typesafeEvaluation
11
+ }
12
+
13
+ export const EVALUATION_PROVIDERS = Object.freeze(Object.keys(EVALUATION_ADAPTERS))
14
+
15
+ /**
16
+ * @param {string} provider
17
+ */
18
+ export function getEvaluationAdapter (provider) {
19
+ const adapter = EVALUATION_ADAPTERS[provider]
20
+ if (!adapter) throw new Error(`no evaluation adapter for provider: ${provider}`)
21
+ return adapter
22
+ }
@@ -0,0 +1,118 @@
1
+ /**
2
+ * TypeSafe evaluation adapter (`POST /v1/systemone`). TypeSafe calls a binary
3
+ * question `noul` and keys binary criteria `true` / `false`; score
4
+ * probabilities come back keyed by level index as strings.
5
+ *
6
+ * @module session/adapters/evaluation/typesafe
7
+ */
8
+
9
+ import { getSpec } from '../_catalog.js'
10
+ import { classifyProviderError, fromHttpStatus, typedError } from '../_errors.js'
11
+ import { evaluationCostFor } from '../_pricing.js'
12
+ import { catalogKey, bareOf } from '#core/model-id.js'
13
+ import { checkEvaluation } from './_shared.js'
14
+
15
+ const BASE_URL = 'https://api.typesafe.ai/v1'
16
+
17
+ export async function typesafeEvaluation (envelope, deps = {}) {
18
+ const fetchFn = deps.fetch ?? globalThis.fetch
19
+ const spec = deps.spec ?? getSpec(catalogKey(envelope.model)) ?? {}
20
+ const start = String(process.hrtime.bigint())
21
+
22
+ checkEvaluation(envelope, spec)
23
+
24
+ const body = {
25
+ model: spec.model ?? bareOf(envelope.model),
26
+ state: envelope.state,
27
+ questions: Object.fromEntries(
28
+ Object.entries(envelope.questions).map(([id, q]) => [id, toQuestion(q)])
29
+ )
30
+ }
31
+
32
+ let res
33
+ try {
34
+ res = await fetchFn(`${BASE_URL}/systemone`, {
35
+ method: 'POST',
36
+ headers: {
37
+ 'Content-Type': 'application/json',
38
+ Authorization: `Bearer ${envelope.auth.key}`
39
+ },
40
+ body: JSON.stringify(body)
41
+ })
42
+ } catch (e) {
43
+ throw typedError(classifyProviderError(e, envelope.auth?.key).message, 'NET_ERROR', true)
44
+ }
45
+
46
+ if (!res.ok) {
47
+ const detail = await res.text().catch(() => '')
48
+ throw fromHttpStatus(res.status, 'evaluation request failed', detail.slice(0, 500), envelope.auth?.key)
49
+ }
50
+
51
+ const payload = await res.json()
52
+ const usage = payload?.usage
53
+ if (typeof payload?.model !== 'string' || !isCount(usage?.input_tokens) || !isCount(usage?.output_tokens)) {
54
+ throw mismatch('response lacks model or token usage')
55
+ }
56
+
57
+ const answers = {}
58
+ for (const [id, q] of Object.entries(envelope.questions)) {
59
+ answers[id] = fromAnswer(id, q, payload.answers?.[id])
60
+ }
61
+
62
+ const inputTokens = usage.input_tokens
63
+ const outputTokens = usage.output_tokens
64
+ const end = String(process.hrtime.bigint())
65
+
66
+ return {
67
+ status: 'completed',
68
+ answers,
69
+ upstreamModel: payload.model,
70
+ inputTokens,
71
+ outputTokens,
72
+ cost: evaluationCostFor(envelope, spec, { inputTokens, outputTokens }),
73
+ timestamps: { start, first: end, end }
74
+ }
75
+ }
76
+
77
+ function toQuestion (q) {
78
+ if (q.type !== 'binary') return q
79
+ const out = { type: 'noul', instructions: q.instructions }
80
+ if (q.criteria) out.criteria = { true: q.criteria.yes, false: q.criteria.no }
81
+ return out
82
+ }
83
+
84
+ function fromAnswer (id, q, a) {
85
+ if (q.type === 'binary') {
86
+ if (a?.type !== 'noul' || !isProbability(a.noul)) throw mismatch(`no binary answer for '${id}'`)
87
+ return { type: 'binary', probability: a.noul }
88
+ }
89
+ if (a?.type !== q.type || !isProbability(a.confidence)) throw mismatch(`no ${q.type} answer for '${id}'`)
90
+
91
+ if (q.type === 'choice') {
92
+ const options = Object.keys(q.criteria)
93
+ if (typeof a.choice !== 'string' || !options.includes(a.choice) ||
94
+ !options.every(o => isProbability(a.probabilities?.[o]))) {
95
+ throw mismatch(`malformed choice answer for '${id}'`)
96
+ }
97
+ const probabilities = Object.fromEntries(options.map(o => [o, a.probabilities[o]]))
98
+ return { type: 'choice', choice: a.choice, probabilities, confidence: a.confidence }
99
+ }
100
+
101
+ const probabilities = q.criteria.map((_, i) => a.probabilities?.[String(i)])
102
+ if (typeof a.score !== 'number' || !probabilities.every(isProbability)) {
103
+ throw mismatch(`malformed score answer for '${id}'`)
104
+ }
105
+ return { type: 'score', score: a.score, probabilities, confidence: a.confidence }
106
+ }
107
+
108
+ function isProbability (v) {
109
+ return typeof v === 'number' && v >= 0 && v <= 1
110
+ }
111
+
112
+ function isCount (v) {
113
+ return Number.isInteger(v) && v >= 0
114
+ }
115
+
116
+ function mismatch (message) {
117
+ return typedError(message, 'EVALUATE_RESULT_MISMATCH', false)
118
+ }
@@ -167,12 +167,12 @@ export async function * openai (envelope, deps = {}) {
167
167
  }
168
168
  break
169
169
 
170
- case 'response.failed':
171
- if (providerOf(envelope.model) === 'chatgpt') {
172
- yield { type: 'error', error: classifyProviderError(new Error('ChatGPT response failed'), envelope.auth?.key, { provider: 'chatgpt' }) }
173
- return
174
- }
175
- break
170
+ case 'response.failed': {
171
+ const provider = providerOf(envelope.model)
172
+ const failure = Object.assign(new Error(`${provider} response failed`), { error: event.response?.error })
173
+ yield { type: 'error', error: classifyProviderError(failure, envelope.auth?.key, { provider }) }
174
+ return
175
+ }
176
176
 
177
177
  default:
178
178
  break
@@ -194,8 +194,9 @@ export async function * openai (envelope, deps = {}) {
194
194
  }
195
195
 
196
196
  const end = String(process.hrtime.bigint())
197
- if (providerOf(envelope.model) === 'chatgpt' && !terminalResponse) {
198
- yield { type: 'error', error: classifyProviderError(new Error('ChatGPT stream ended before completion'), envelope.auth?.key, { provider: 'chatgpt' }) }
197
+ if (!terminalResponse) {
198
+ const provider = providerOf(envelope.model)
199
+ yield { type: 'error', error: classifyProviderError(new Error(`${provider} stream ended before completion`), envelope.auth?.key, { provider }) }
199
200
  return
200
201
  }
201
202
  // OpenAI Responses reports `output_tokens` INCLUDING reasoning
@@ -19,6 +19,7 @@ import { run } from './run.js'
19
19
  import { runImage } from './run_image.js'
20
20
  import { runTranscription } from './run_transcription.js'
21
21
  import { runEmbedding } from './run_embedding.js'
22
+ import { runEvaluation } from './run_evaluation.js'
22
23
  import { runInfo } from './run_info.js'
23
24
  import { setCatalog } from './adapters/_catalog.js'
24
25
 
@@ -233,6 +234,14 @@ export async function drive (stdin, stdout) {
233
234
  } else {
234
235
  await writeLine({ type: 'error', error: out.error })
235
236
  }
237
+ } else if (envelope.op === 'evaluate') {
238
+ const { op: _op, ...evEnv } = envelope
239
+ const out = await runEvaluation(evEnv)
240
+ if (out.ok) {
241
+ await writeLine({ type: 'evaluate_done', result: out.result })
242
+ } else {
243
+ await writeLine({ type: 'error', error: out.error })
244
+ }
236
245
  } else if (envelope.op === 'info') {
237
246
  const out = await runInfo(envelope)
238
247
  if (out.ok) {
package/js/session/run.js CHANGED
@@ -23,6 +23,7 @@
23
23
  */
24
24
 
25
25
  import { ADAPTER_NAMES, isImageProvider } from './adapters/_registry.js'
26
+ import { EVALUATION_PROVIDERS } from './adapters/evaluation/index.js'
26
27
  import { getSpec } from './adapters/_catalog.js'
27
28
  import { getProviderLimits } from './adapters/_providers.js'
28
29
  import { hasSpeed, mergeSpeed, speedHasOwnQuota, speedNames } from './adapters/_speed.js'
@@ -119,6 +120,13 @@ export async function * run (envelope, {
119
120
  yield err
120
121
  return
121
122
  }
123
+ if (EVALUATION_PROVIDERS.includes(provider)) {
124
+ const detail = `provider '${provider}' supports evaluation only; use evaluate(...) instead`
125
+ log.warn({ provider }, '[mohdel:answer] evaluation-only provider via answer')
126
+ endSpanError(span, new Error(detail))
127
+ yield errorEvent(detail, 'PROVIDER_TEXT_NOT_SUPPORTED')
128
+ return
129
+ }
122
130
  const err = errorEvent(messageOf(e), 'SESSION_UNKNOWN_PROVIDER')
123
131
  log.warn({ err: e, provider }, '[mohdel:answer] unknown provider')
124
132
  endSpanError(span, e)
@@ -0,0 +1,84 @@
1
+ /**
2
+ * Evaluation runtime. Resolves the adapter for the envelope's provider and
3
+ * returns either a result or a typed error, never throwing. Rate limits as in
4
+ * `run_embedding.js`, minus the input count.
5
+ *
6
+ * @module session/run_evaluation
7
+ */
8
+
9
+ import { getEvaluationAdapter } from './adapters/evaluation/index.js'
10
+ import { classifyProviderError } from './adapters/_errors.js'
11
+ import { getProviderLimits } from './adapters/_providers.js'
12
+ import * as defaultLimiter from './_rate_limiter.js'
13
+ import { providerOf } from '#core/model-id.js'
14
+
15
+ /**
16
+ * @param {import('#core/evaluation.js').EvaluateEnvelope} envelope
17
+ * @param {{
18
+ * resolveAdapter?: (provider: string) => any,
19
+ * resolveProviderLimits?: (provider: string) => any,
20
+ * limiter?: any,
21
+ * sleep?: (ms: number) => Promise<void>,
22
+ * modelKey?: string,
23
+ * spec?: any
24
+ * }} [options]
25
+ * @returns {Promise<
26
+ * | {ok: true, result: import('#core/evaluation.js').EvaluateResult}
27
+ * | {ok: false, error: import('#core/errors.js').TypedError}
28
+ * >}
29
+ */
30
+ export async function runEvaluation (envelope, {
31
+ resolveAdapter = getEvaluationAdapter,
32
+ resolveProviderLimits = getProviderLimits,
33
+ limiter = defaultLimiter,
34
+ sleep = defaultSleep,
35
+ modelKey = envelope.model,
36
+ spec
37
+ } = {}) {
38
+ const provider = providerOf(envelope.model)
39
+
40
+ let adapter
41
+ try {
42
+ adapter = resolveAdapter(provider)
43
+ } catch (e) {
44
+ return {
45
+ ok: false,
46
+ error: {
47
+ message: messageOf(e),
48
+ severity: 'error',
49
+ retryable: false,
50
+ type: 'SESSION_UNKNOWN_PROVIDER'
51
+ }
52
+ }
53
+ }
54
+
55
+ const providerCfg = resolveProviderLimits(provider) || {}
56
+ const rpmLimit = spec?.rpmLimit ?? providerCfg.rpmLimit
57
+ const tpmLimit = spec?.tpmLimit ?? providerCfg.tpmLimit
58
+ const bucketKey = spec?.rateLimitScope === 'model' ? modelKey : provider
59
+
60
+ if (rpmLimit != null || tpmLimit != null) {
61
+ const delay = limiter.check(bucketKey, { rpmLimit, tpmLimit })
62
+ if (delay > 0) await sleep(delay)
63
+ limiter.recordRequest(bucketKey)
64
+ }
65
+
66
+ try {
67
+ const result = await adapter(envelope, spec ? { spec } : {})
68
+ if (tpmLimit != null) limiter.recordTokens(bucketKey, result.inputTokens + result.outputTokens)
69
+ return { ok: true, result }
70
+ } catch (e) {
71
+ const typed = /** @type {any} */(e).typed || classifyProviderError(e, envelope.auth?.key)
72
+ return { ok: false, error: typed }
73
+ }
74
+ }
75
+
76
+ /** @param {unknown} e */
77
+ function messageOf (e) {
78
+ return e instanceof Error ? e.message : String(e)
79
+ }
80
+
81
+ /** @param {number} ms */
82
+ function defaultSleep (ms) {
83
+ return new Promise(resolve => setTimeout(resolve, ms))
84
+ }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "mohdel",
3
- "version": "3.7.0",
3
+ "version": "3.8.1",
4
4
  "license": "MIT",
5
5
  "author": {
6
6
  "name": "Christophe Le Bars",
@@ -144,7 +144,7 @@
144
144
  "@opentelemetry/exporter-trace-otlp-grpc": "^0.222.0",
145
145
  "@opentelemetry/sdk-node": "^0.222.0",
146
146
  "chalk": "^6.0.1",
147
- "mohdel-thin-gate-linux-x64-gnu": "3.7.0"
147
+ "mohdel-thin-gate-linux-x64-gnu": "3.8.1"
148
148
  },
149
149
  "dependencies": {
150
150
  "@anthropic-ai/sdk": "^0.129.0",
@@ -65,6 +65,12 @@ const creators = {
65
65
  logo: 'cohere.svg',
66
66
  description: 'Cohere builds retrieval-focused models: embeddings and rerankers aimed at enterprise search rather than chat.'
67
67
  },
68
+ typesafe: {
69
+ prefixes: ['jev'],
70
+ label: 'TypeSafe',
71
+ logo: 'typesafe.svg',
72
+ description: 'TypeSafe trains decision models that answer typed questions with calibrated probabilities instead of generating text.'
73
+ },
68
74
  nomic: {
69
75
  prefixes: ['nomic-embed'],
70
76
  label: 'Nomic',
package/src/lib/index.js CHANGED
@@ -16,7 +16,7 @@ import {
16
16
  import { createRateLimiter } from '../../js/session/_rate_limiter.js'
17
17
  import { createCooldownTracker } from '../../js/session/_cooldown.js'
18
18
  import { setCatalog } from '../../js/session/adapters/_catalog.js'
19
- import { runAnswer, runAnswerEmbedding, runAnswerImage, runAnswerTranscription } from '../../js/factory/bridge.js'
19
+ import { runAnswer, runAnswerEmbedding, runAnswerEvaluation, runAnswerImage, runAnswerTranscription } from '../../js/factory/bridge.js'
20
20
  import { startSpan, endSpanOk, endSpanError } from './tracing.js'
21
21
  import { isValidTag } from './schema.js'
22
22
  import { silent } from './logger.js'
@@ -751,6 +751,22 @@ const createModelProxy = (resolvedModelId, modelSpec, handlers, aliasOutputEffor
751
751
  }
752
752
  }
753
753
 
754
+ if (prop === 'evaluate') {
755
+ return async (state, questions, options = {}) => {
756
+ const { configuration } = await getRuntime()
757
+ return runAnswerEvaluation({
758
+ provider: modelSpec.provider,
759
+ model: modelSpec.model ?? resolvedModelId.split('/').pop(),
760
+ modelKey: resolvedModelId,
761
+ configuration,
762
+ state,
763
+ questions,
764
+ options,
765
+ spec: modelSpec
766
+ }, { limiter: rateLimiter, resolveProviderLimits })
767
+ }
768
+ }
769
+
754
770
  if (prop === 'setRateLimit') {
755
771
  return async ({ rpm, tpm, inpm } = {}) => {
756
772
  const curatedCache = getCuratedCacheSnapshot()
@@ -100,6 +100,13 @@ const PROVIDER_INFO = {
100
100
  hint: 'Create an API key in the Cohere dashboard under API Keys',
101
101
  free: true
102
102
  },
103
+ typesafe: {
104
+ label: 'TypeSafe',
105
+ description: 'Calibrated yes/no, choice and score answers about a state. No chat models through mohdel.',
106
+ url: 'https://console.typesafe.ai/keys',
107
+ hint: 'Create an API key in the TypeSafe console under Keys',
108
+ free: false
109
+ },
103
110
  qwen: {
104
111
  label: 'Qwen Cloud',
105
112
  description: 'Qwen — reasoning, coding, long context. Free quota for new users.',
@@ -144,6 +144,20 @@ const providers = {
144
144
  rateLimits: 'https://docs.cohere.com/docs/rate-limits'
145
145
  }
146
146
  },
147
+ typesafe: {
148
+ billing: { kind: 'metered' },
149
+ sdk: 'typesafe',
150
+ api: 'evaluation',
151
+ catalog: false,
152
+ apiKeyEnv: 'TYPESAFE_API_SK',
153
+ baseURL: 'https://api.typesafe.ai/v1',
154
+ createConfiguration: apiKey => ({ apiKey }),
155
+ references: {
156
+ pricing: 'https://docs.typesafe.ai/models',
157
+ models: 'https://docs.typesafe.ai/models',
158
+ rateLimits: 'https://docs.typesafe.ai/models'
159
+ }
160
+ },
147
161
  mistral: {
148
162
  billing: { kind: 'metered' },
149
163
  sdk: 'openai',
package/src/lib/schema.js CHANGED
@@ -65,6 +65,7 @@ const fieldDefs = {
65
65
  maxInputTokens: { type: 'number' },
66
66
  inputTypes: { type: 'object' },
67
67
  defaultInputType: { type: 'string' },
68
+ evaluationTypes: { type: 'array', itemType: 'string' },
68
69
  deprecated: { type: 'string' },
69
70
  suspended: { type: 'string' },
70
71
  rpmLimit: { type: 'number' },