mohdel 3.7.0 → 3.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -0
- package/config/curated.schema.json +8 -2
- package/js/client/call_evaluation.js +47 -0
- package/js/client/index.js +1 -0
- package/js/core/evaluation.js +105 -0
- package/js/factory/bridge.js +39 -0
- package/js/session/adapters/_errors.js +4 -2
- package/js/session/adapters/_pricing.js +10 -0
- package/js/session/adapters/embedding/cohere.js +1 -1
- package/js/session/adapters/embedding/gemini.js +1 -1
- package/js/session/adapters/embedding/openai_compatible.js +1 -1
- package/js/session/adapters/evaluation/_shared.js +73 -0
- package/js/session/adapters/evaluation/index.js +22 -0
- package/js/session/adapters/evaluation/typesafe.js +118 -0
- package/js/session/driver.js +9 -0
- package/js/session/run.js +8 -0
- package/js/session/run_evaluation.js +84 -0
- package/package.json +2 -2
- package/src/lib/creators.js +6 -0
- package/src/lib/index.js +17 -1
- package/src/lib/provider-info.js +7 -0
- package/src/lib/providers.js +14 -0
- package/src/lib/schema.js +1 -0
package/README.md
CHANGED
|
@@ -472,6 +472,7 @@ QWEN_API_SK=sk-...
|
|
|
472
472
|
XIAOMI_API_SK=...
|
|
473
473
|
META_API_SK=...
|
|
474
474
|
COHERE_API_SK=...
|
|
475
|
+
TYPESAFE_API_SK=...
|
|
475
476
|
MOHDEL_LOCAL_API_SK=...
|
|
476
477
|
```
|
|
477
478
|
|
|
@@ -515,6 +516,7 @@ What each provider supports through mohdel's unified interface:
|
|
|
515
516
|
| OpenRouter | Yes | Yes | Yes | No | Varies | Meta-provider; `providerOptions.openrouter` for routing prefs |
|
|
516
517
|
| Local | Yes | Yes | Yes | No | No | Any OpenAI-compatible server; endpoint is the catalog entry's `baseURL` |
|
|
517
518
|
| Cohere | n/a | n/a | n/a | n/a | n/a | Embeddings only: no chat models reach mohdel through it |
|
|
519
|
+
| TypeSafe | n/a | n/a | n/a | n/a | n/a | Evaluation only (`evaluate()`): typed questions answered with probabilities |
|
|
518
520
|
| Novita | Yes | Yes | Yes | No | Yes (`reasoning_content`) | Prices in the model list; text via the shared chat-completions path, separate image adapter |
|
|
519
521
|
|
|
520
522
|
Adapter capability ≠ model capability — whether a given model accepts images, tools, or thinking effort depends on the model spec in `curated.json`. The adapter passes through what the envelope supplies; the provider rejects unsupported combos.
|
|
@@ -87,10 +87,11 @@
|
|
|
87
87
|
"enum": [
|
|
88
88
|
"model",
|
|
89
89
|
"image",
|
|
90
|
-
"transcription"
|
|
90
|
+
"transcription",
|
|
91
|
+
"evaluation"
|
|
91
92
|
],
|
|
92
93
|
"default": "model",
|
|
93
|
-
"description": "'model' for chat/completion, 'image' for image generation, 'transcription' for speech-to-text."
|
|
94
|
+
"description": "'model' for chat/completion, 'image' for image generation, 'transcription' for speech-to-text, 'evaluation' for typed-question answering (evaluate())."
|
|
94
95
|
},
|
|
95
96
|
"label": {
|
|
96
97
|
"type": "string",
|
|
@@ -550,6 +551,11 @@
|
|
|
550
551
|
"type": "string",
|
|
551
552
|
"description": "Symbolic role sent when the caller names none. Required for providers that make the parameter mandatory (Cohere v3+)."
|
|
552
553
|
},
|
|
554
|
+
"evaluationTypes": {
|
|
555
|
+
"type": "array",
|
|
556
|
+
"items": { "type": "string", "enum": ["binary", "choice", "score"] },
|
|
557
|
+
"description": "Question types an evaluation model answers. A question of another type fails before dispatch. Without it, every type is sent and the provider decides."
|
|
558
|
+
},
|
|
553
559
|
"rpmLimit": {
|
|
554
560
|
"type": "integer",
|
|
555
561
|
"minimum": 1,
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Send an EvaluateEnvelope to thin-gate's `POST /v1/evaluate`. One-shot: a
|
|
3
|
+
* single JSON response body.
|
|
4
|
+
*
|
|
5
|
+
* @module client/call_evaluation
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
import { requestUnix } from './transport.js'
|
|
9
|
+
import { readAll, parseErrorBody } from './response.js'
|
|
10
|
+
import { MohdelError } from '#core'
|
|
11
|
+
|
|
12
|
+
/**
|
|
13
|
+
* @param {import('#core/evaluation.js').EvaluateEnvelope} envelope
|
|
14
|
+
* @param {object} options
|
|
15
|
+
* @param {string} options.socketPath
|
|
16
|
+
* @param {AbortSignal} [options.signal]
|
|
17
|
+
* @param {string} [options.path] HTTP path; defaults to '/v1/evaluate'
|
|
18
|
+
* @param {Record<string, string>} [options.headers] sent with the request, for a router in front of the gate;
|
|
19
|
+
* `content-type`, `content-length`, `transfer-encoding`, `connection` and `host` are the transport's
|
|
20
|
+
* @returns {Promise<import('#core/evaluation.js').EvaluateResult>}
|
|
21
|
+
*/
|
|
22
|
+
export async function callEvaluation (envelope, { socketPath, signal, path = '/v1/evaluate', headers }) {
|
|
23
|
+
const res = await requestUnix({
|
|
24
|
+
socketPath,
|
|
25
|
+
path,
|
|
26
|
+
method: 'POST',
|
|
27
|
+
body: envelope,
|
|
28
|
+
signal,
|
|
29
|
+
headers
|
|
30
|
+
})
|
|
31
|
+
|
|
32
|
+
const body = await readAll(res)
|
|
33
|
+
|
|
34
|
+
if (res.statusCode !== 200) {
|
|
35
|
+
throw MohdelError.fromJSON(parseErrorBody(body, res.statusCode ?? 0))
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
try {
|
|
39
|
+
return JSON.parse(body)
|
|
40
|
+
} catch (e) {
|
|
41
|
+
throw new MohdelError(`gate returned an unparseable evaluate body: ${body.slice(0, 200)}`, {
|
|
42
|
+
type: 'PROTOCOL_INVALID_RESPONSE',
|
|
43
|
+
severity: 'error',
|
|
44
|
+
retryable: false
|
|
45
|
+
})
|
|
46
|
+
}
|
|
47
|
+
}
|
package/js/client/index.js
CHANGED
|
@@ -18,4 +18,5 @@ export { coalesce } from './coalesce.js'
|
|
|
18
18
|
export { callImage } from './call_image.js'
|
|
19
19
|
export { callTranscription } from './call_transcription.js'
|
|
20
20
|
export { callEmbedding } from './call_embedding.js'
|
|
21
|
+
export { callEvaluation } from './call_evaluation.js'
|
|
21
22
|
export { resolveGateBinary } from './gate-binary.js'
|
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Evaluation envelope and result: one `state`, a map of typed questions, one
|
|
3
|
+
* typed answer per question.
|
|
4
|
+
*
|
|
5
|
+
* @module core/evaluation
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
/**
|
|
9
|
+
* @typedef {object} EvaluateEnvelope
|
|
10
|
+
* @property {string} callId
|
|
11
|
+
* @property {string} authId
|
|
12
|
+
* @property {import('./envelope.js').Auth} auth
|
|
13
|
+
* @property {string} [traceparent]
|
|
14
|
+
* @property {string} [baggage]
|
|
15
|
+
* @property {string} model
|
|
16
|
+
* @property {string | object | any[]} state
|
|
17
|
+
* @property {Record<string, Question>} questions
|
|
18
|
+
* Answers come back under the same ids.
|
|
19
|
+
*/
|
|
20
|
+
|
|
21
|
+
/**
|
|
22
|
+
* @typedef {BinaryQuestion | ChoiceQuestion | ScoreQuestion} Question
|
|
23
|
+
*
|
|
24
|
+
* `instructions` and every criterion description may be a string, an object
|
|
25
|
+
* or an array.
|
|
26
|
+
*/
|
|
27
|
+
|
|
28
|
+
/**
|
|
29
|
+
* @typedef {object} BinaryQuestion
|
|
30
|
+
* @property {'binary'} type
|
|
31
|
+
* @property {any} instructions
|
|
32
|
+
* @property {{yes: any, no: any}} [criteria]
|
|
33
|
+
*/
|
|
34
|
+
|
|
35
|
+
/**
|
|
36
|
+
* @typedef {object} ChoiceQuestion
|
|
37
|
+
* @property {'choice'} type
|
|
38
|
+
* @property {any} instructions
|
|
39
|
+
* @property {Record<string, any>} criteria
|
|
40
|
+
* Option name to its description, or null. At least two options.
|
|
41
|
+
*/
|
|
42
|
+
|
|
43
|
+
/**
|
|
44
|
+
* @typedef {object} ScoreQuestion
|
|
45
|
+
* @property {'score'} type
|
|
46
|
+
* @property {any} instructions
|
|
47
|
+
* @property {any[]} criteria
|
|
48
|
+
* Level descriptions, lowest first; level `i` is worth `i`. At least two.
|
|
49
|
+
*/
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* @typedef {BinaryAnswer | ChoiceAnswer | ScoreAnswer} Answer
|
|
53
|
+
*/
|
|
54
|
+
|
|
55
|
+
/**
|
|
56
|
+
* @typedef {object} BinaryAnswer
|
|
57
|
+
* @property {'binary'} type
|
|
58
|
+
* @property {number} probability
|
|
59
|
+
* Of yes.
|
|
60
|
+
*/
|
|
61
|
+
|
|
62
|
+
/**
|
|
63
|
+
* @typedef {object} ChoiceAnswer
|
|
64
|
+
* @property {'choice'} type
|
|
65
|
+
* @property {string} choice
|
|
66
|
+
* @property {Record<string, number>} probabilities
|
|
67
|
+
* @property {number} [confidence]
|
|
68
|
+
* Absent when the provider reports none.
|
|
69
|
+
*/
|
|
70
|
+
|
|
71
|
+
/**
|
|
72
|
+
* @typedef {object} ScoreAnswer
|
|
73
|
+
* @property {'score'} type
|
|
74
|
+
* @property {number} score
|
|
75
|
+
* Probability-weighted level; can fall between two.
|
|
76
|
+
* @property {number[]} probabilities
|
|
77
|
+
* One per level, in `criteria` order.
|
|
78
|
+
* @property {number} [confidence]
|
|
79
|
+
*/
|
|
80
|
+
|
|
81
|
+
/**
|
|
82
|
+
* @typedef {object} EvaluateResult
|
|
83
|
+
* @property {'completed'} status
|
|
84
|
+
* @property {Record<string, Answer>} answers
|
|
85
|
+
* @property {string} upstreamModel
|
|
86
|
+
* The provider's id for the version that answered, which differs from the
|
|
87
|
+
* entry's `model` when that is an alias.
|
|
88
|
+
* @property {number} inputTokens
|
|
89
|
+
* @property {number} outputTokens
|
|
90
|
+
* @property {number} cost
|
|
91
|
+
* @property {{start: string, first: string, end: string}} timestamps
|
|
92
|
+
*/
|
|
93
|
+
|
|
94
|
+
export const EVALUATE_ENVELOPE_FIELDS = Object.freeze([
|
|
95
|
+
'callId',
|
|
96
|
+
'authId',
|
|
97
|
+
'auth',
|
|
98
|
+
'traceparent',
|
|
99
|
+
'baggage',
|
|
100
|
+
'model',
|
|
101
|
+
'state',
|
|
102
|
+
'questions'
|
|
103
|
+
])
|
|
104
|
+
|
|
105
|
+
export const QUESTION_TYPES = Object.freeze(['binary', 'choice', 'score'])
|
package/js/factory/bridge.js
CHANGED
|
@@ -24,6 +24,7 @@ import { run } from '../session/run.js'
|
|
|
24
24
|
import { runImage } from '../session/run_image.js'
|
|
25
25
|
import { runTranscription } from '../session/run_transcription.js'
|
|
26
26
|
import { runEmbedding } from '../session/run_embedding.js'
|
|
27
|
+
import { runEvaluation } from '../session/run_evaluation.js'
|
|
27
28
|
import { markTrustedMedia } from '../session/adapters/_media.js'
|
|
28
29
|
import { MohdelError, validateIds } from '#core'
|
|
29
30
|
import { createRealtimeDeltaBuffer } from '../../src/lib/utils.js'
|
|
@@ -224,6 +225,44 @@ export async function runAnswerEmbedding ({ provider, model, modelKey, configura
|
|
|
224
225
|
return out.result
|
|
225
226
|
}
|
|
226
227
|
|
|
228
|
+
/**
|
|
229
|
+
* Run an `evaluate()` call through the /session runtime.
|
|
230
|
+
*
|
|
231
|
+
* @param {object} args
|
|
232
|
+
* @param {string} args.provider
|
|
233
|
+
* @param {string} args.model
|
|
234
|
+
* @param {string} [args.modelKey]
|
|
235
|
+
* @param {any} args.configuration
|
|
236
|
+
* @param {import('#core/evaluation.js').EvaluateEnvelope['state']} args.state
|
|
237
|
+
* @param {import('#core/evaluation.js').EvaluateEnvelope['questions']} args.questions
|
|
238
|
+
* @param {any} [args.options] `callId` / `authId` only.
|
|
239
|
+
* @param {any} [args.spec]
|
|
240
|
+
* @param {BridgeDeps} [deps]
|
|
241
|
+
* @returns {Promise<import('#core/evaluation.js').EvaluateResult>}
|
|
242
|
+
*/
|
|
243
|
+
export async function runAnswerEvaluation ({ provider, model, modelKey, configuration, state, questions, options = {}, spec }, deps = {}) {
|
|
244
|
+
const callId = options.callId || newCallId()
|
|
245
|
+
const authId = options.authId || 'local'
|
|
246
|
+
assertValidIds(callId, authId, `${provider}/${model}`)
|
|
247
|
+
|
|
248
|
+
const envelope = {
|
|
249
|
+
callId,
|
|
250
|
+
authId,
|
|
251
|
+
auth: configToAuth(configuration),
|
|
252
|
+
model: `${provider}/${model}`,
|
|
253
|
+
state,
|
|
254
|
+
questions
|
|
255
|
+
}
|
|
256
|
+
|
|
257
|
+
const out = await runEvaluation(envelope, {
|
|
258
|
+
...deps,
|
|
259
|
+
...(modelKey ? { modelKey } : {}),
|
|
260
|
+
...(spec ? { spec } : {})
|
|
261
|
+
})
|
|
262
|
+
if (!out.ok) throw MohdelError.fromJSON(out.error, { provider, model })
|
|
263
|
+
return out.result
|
|
264
|
+
}
|
|
265
|
+
|
|
227
266
|
/**
|
|
228
267
|
* @param {object} args
|
|
229
268
|
* @param {string} args.modelKey Mohdel catalog key `<provider>/<bare>`. The
|
|
@@ -469,9 +469,11 @@ export function typedError (message, type, retryable, detail) {
|
|
|
469
469
|
* @param {number} status
|
|
470
470
|
* @param {string} message
|
|
471
471
|
* @param {string} [detail]
|
|
472
|
+
* @param {string} [key] masked out of `detail`
|
|
472
473
|
* @returns {Error & {typed: import('#core/errors.js').TypedError}}
|
|
473
474
|
*/
|
|
474
|
-
export function fromHttpStatus (status, message, detail) {
|
|
475
|
+
export function fromHttpStatus (status, message, detail, key) {
|
|
475
476
|
const typed = classifyProviderError({ status })
|
|
476
|
-
|
|
477
|
+
const scrubbed = scrubKey(detail, key)
|
|
478
|
+
return typedError(typed.message, typed.type, typed.retryable, scrubbed ? `${message}: ${scrubbed}` : message)
|
|
477
479
|
}
|
|
@@ -146,6 +146,16 @@ export function embeddingCostFor (envelope, spec, usage) {
|
|
|
146
146
|
return unmetered(envelope.model) ? 0 : computeEmbeddingCost(spec, usage)
|
|
147
147
|
}
|
|
148
148
|
|
|
149
|
+
/**
|
|
150
|
+
* @param {{model: string}} envelope
|
|
151
|
+
* @param {any} spec
|
|
152
|
+
* @param {{inputTokens: number, outputTokens: number}} usage
|
|
153
|
+
* @returns {number}
|
|
154
|
+
*/
|
|
155
|
+
export function evaluationCostFor (envelope, spec, usage) {
|
|
156
|
+
return unmetered(envelope.model) ? 0 : computeCost(spec, usage)
|
|
157
|
+
}
|
|
158
|
+
|
|
149
159
|
/**
|
|
150
160
|
* Cost of a transcription call.
|
|
151
161
|
*
|
|
@@ -61,7 +61,7 @@ export async function cohereEmbedding (envelope, deps = {}) {
|
|
|
61
61
|
|
|
62
62
|
if (!res.ok) {
|
|
63
63
|
const detail = await res.text().catch(() => '')
|
|
64
|
-
throw fromHttpStatus(res.status, detail, envelope.auth?.key)
|
|
64
|
+
throw fromHttpStatus(res.status, 'embedding request failed', detail.slice(0, 500), envelope.auth?.key)
|
|
65
65
|
}
|
|
66
66
|
|
|
67
67
|
const payload = await res.json()
|
|
@@ -59,7 +59,7 @@ export async function geminiEmbedding (envelope, deps = {}) {
|
|
|
59
59
|
|
|
60
60
|
if (!res.ok) {
|
|
61
61
|
const detail = await res.text().catch(() => '')
|
|
62
|
-
throw fromHttpStatus(res.status, detail, envelope.auth?.key)
|
|
62
|
+
throw fromHttpStatus(res.status, 'embedding request failed', detail.slice(0, 500), envelope.auth?.key)
|
|
63
63
|
}
|
|
64
64
|
|
|
65
65
|
const payload = await res.json()
|
|
@@ -56,7 +56,7 @@ export function createEmbeddingAdapter ({ baseURL, dimensionsField = 'dimensions
|
|
|
56
56
|
|
|
57
57
|
if (!res.ok) {
|
|
58
58
|
const detail = await res.text().catch(() => '')
|
|
59
|
-
throw fromHttpStatus(res.status, detail, envelope.auth?.key)
|
|
59
|
+
throw fromHttpStatus(res.status, 'embedding request failed', detail.slice(0, 500), envelope.auth?.key)
|
|
60
60
|
}
|
|
61
61
|
|
|
62
62
|
const payload = await res.json()
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Pre-dispatch checks every evaluation adapter runs. They reject what the
|
|
3
|
+
* gate's serde types reject, so a malformed envelope fails the same way
|
|
4
|
+
* in-process.
|
|
5
|
+
*
|
|
6
|
+
* @module session/adapters/evaluation/shared
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
import { typedError } from '../_errors.js'
|
|
10
|
+
import { QUESTION_TYPES } from '#core/evaluation.js'
|
|
11
|
+
|
|
12
|
+
const QUESTION_KEYS = new Set(['type', 'instructions', 'criteria'])
|
|
13
|
+
const BINARY_CRITERIA_KEYS = ['yes', 'no']
|
|
14
|
+
|
|
15
|
+
/**
|
|
16
|
+
* @param {import('#core/evaluation.js').EvaluateEnvelope} envelope
|
|
17
|
+
* @param {any} spec
|
|
18
|
+
*/
|
|
19
|
+
export function checkEvaluation (envelope, spec) {
|
|
20
|
+
const { state, questions } = envelope
|
|
21
|
+
if (typeof state !== 'string' && !isObject(state) && !Array.isArray(state)) {
|
|
22
|
+
throw invalid('state must be a string, an object or an array')
|
|
23
|
+
}
|
|
24
|
+
if (!isObject(questions) || Object.keys(questions).length === 0) {
|
|
25
|
+
throw invalid('questions must be a non-empty object')
|
|
26
|
+
}
|
|
27
|
+
for (const [id, q] of Object.entries(questions)) {
|
|
28
|
+
checkQuestion(id, q, spec)
|
|
29
|
+
}
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
function checkQuestion (id, q, spec) {
|
|
33
|
+
if (!isObject(q)) throw invalid(`question '${id}' must be an object`)
|
|
34
|
+
const extra = Object.keys(q).filter(k => !QUESTION_KEYS.has(k))
|
|
35
|
+
if (extra.length) throw invalid(`question '${id}' has unknown field '${extra.join("', '")}'`)
|
|
36
|
+
if (!QUESTION_TYPES.includes(q.type)) {
|
|
37
|
+
throw invalid(`question '${id}' type must be one of ${QUESTION_TYPES.join(', ')}`)
|
|
38
|
+
}
|
|
39
|
+
if (Array.isArray(spec?.evaluationTypes) && !spec.evaluationTypes.includes(q.type)) {
|
|
40
|
+
throw typedError(
|
|
41
|
+
`question '${id}' is '${q.type}', which this model does not answer; it answers ${spec.evaluationTypes.join(', ')}`,
|
|
42
|
+
'EVALUATE_QUESTION_TYPE_UNSUPPORTED',
|
|
43
|
+
false
|
|
44
|
+
)
|
|
45
|
+
}
|
|
46
|
+
if (q.instructions === undefined || q.instructions === null) {
|
|
47
|
+
throw invalid(`question '${id}' requires instructions`)
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
const c = q.criteria
|
|
51
|
+
if (q.type === 'binary') {
|
|
52
|
+
if (c === undefined) return
|
|
53
|
+
if (!isObject(c) || Object.keys(c).length !== 2 || !BINARY_CRITERIA_KEYS.every(k => c[k] !== undefined && c[k] !== null)) {
|
|
54
|
+
throw invalid(`question '${id}' criteria must have exactly 'yes' and 'no'`)
|
|
55
|
+
}
|
|
56
|
+
} else if (q.type === 'choice') {
|
|
57
|
+
if (!isObject(c) || Object.keys(c).length < 2) {
|
|
58
|
+
throw invalid(`question '${id}' criteria must name at least two options`)
|
|
59
|
+
}
|
|
60
|
+
} else if (!Array.isArray(c) || c.length < 2) {
|
|
61
|
+
throw invalid(`question '${id}' criteria must list at least two levels`)
|
|
62
|
+
}
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
/** @param {unknown} v */
|
|
66
|
+
function isObject (v) {
|
|
67
|
+
return typeof v === 'object' && v !== null && !Array.isArray(v)
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
/** @param {string} message */
|
|
71
|
+
function invalid (message) {
|
|
72
|
+
return typedError(message, 'EVALUATE_INPUT_INVALID', false)
|
|
73
|
+
}
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Evaluation-adapter registry.
|
|
3
|
+
*
|
|
4
|
+
* @module session/adapters/evaluation
|
|
5
|
+
*/
|
|
6
|
+
|
|
7
|
+
import { typesafeEvaluation } from './typesafe.js'
|
|
8
|
+
|
|
9
|
+
const EVALUATION_ADAPTERS = {
|
|
10
|
+
typesafe: typesafeEvaluation
|
|
11
|
+
}
|
|
12
|
+
|
|
13
|
+
export const EVALUATION_PROVIDERS = Object.freeze(Object.keys(EVALUATION_ADAPTERS))
|
|
14
|
+
|
|
15
|
+
/**
|
|
16
|
+
* @param {string} provider
|
|
17
|
+
*/
|
|
18
|
+
export function getEvaluationAdapter (provider) {
|
|
19
|
+
const adapter = EVALUATION_ADAPTERS[provider]
|
|
20
|
+
if (!adapter) throw new Error(`no evaluation adapter for provider: ${provider}`)
|
|
21
|
+
return adapter
|
|
22
|
+
}
|
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* TypeSafe evaluation adapter (`POST /v1/systemone`). TypeSafe calls a binary
|
|
3
|
+
* question `noul` and keys binary criteria `true` / `false`; score
|
|
4
|
+
* probabilities come back keyed by level index as strings.
|
|
5
|
+
*
|
|
6
|
+
* @module session/adapters/evaluation/typesafe
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
import { getSpec } from '../_catalog.js'
|
|
10
|
+
import { classifyProviderError, fromHttpStatus, typedError } from '../_errors.js'
|
|
11
|
+
import { evaluationCostFor } from '../_pricing.js'
|
|
12
|
+
import { catalogKey, bareOf } from '#core/model-id.js'
|
|
13
|
+
import { checkEvaluation } from './_shared.js'
|
|
14
|
+
|
|
15
|
+
const BASE_URL = 'https://api.typesafe.ai/v1'
|
|
16
|
+
|
|
17
|
+
export async function typesafeEvaluation (envelope, deps = {}) {
|
|
18
|
+
const fetchFn = deps.fetch ?? globalThis.fetch
|
|
19
|
+
const spec = deps.spec ?? getSpec(catalogKey(envelope.model)) ?? {}
|
|
20
|
+
const start = String(process.hrtime.bigint())
|
|
21
|
+
|
|
22
|
+
checkEvaluation(envelope, spec)
|
|
23
|
+
|
|
24
|
+
const body = {
|
|
25
|
+
model: spec.model ?? bareOf(envelope.model),
|
|
26
|
+
state: envelope.state,
|
|
27
|
+
questions: Object.fromEntries(
|
|
28
|
+
Object.entries(envelope.questions).map(([id, q]) => [id, toQuestion(q)])
|
|
29
|
+
)
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
let res
|
|
33
|
+
try {
|
|
34
|
+
res = await fetchFn(`${BASE_URL}/systemone`, {
|
|
35
|
+
method: 'POST',
|
|
36
|
+
headers: {
|
|
37
|
+
'Content-Type': 'application/json',
|
|
38
|
+
Authorization: `Bearer ${envelope.auth.key}`
|
|
39
|
+
},
|
|
40
|
+
body: JSON.stringify(body)
|
|
41
|
+
})
|
|
42
|
+
} catch (e) {
|
|
43
|
+
throw typedError(classifyProviderError(e, envelope.auth?.key).message, 'NET_ERROR', true)
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
if (!res.ok) {
|
|
47
|
+
const detail = await res.text().catch(() => '')
|
|
48
|
+
throw fromHttpStatus(res.status, 'evaluation request failed', detail.slice(0, 500), envelope.auth?.key)
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
const payload = await res.json()
|
|
52
|
+
const usage = payload?.usage
|
|
53
|
+
if (typeof payload?.model !== 'string' || !isCount(usage?.input_tokens) || !isCount(usage?.output_tokens)) {
|
|
54
|
+
throw mismatch('response lacks model or token usage')
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
const answers = {}
|
|
58
|
+
for (const [id, q] of Object.entries(envelope.questions)) {
|
|
59
|
+
answers[id] = fromAnswer(id, q, payload.answers?.[id])
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
const inputTokens = usage.input_tokens
|
|
63
|
+
const outputTokens = usage.output_tokens
|
|
64
|
+
const end = String(process.hrtime.bigint())
|
|
65
|
+
|
|
66
|
+
return {
|
|
67
|
+
status: 'completed',
|
|
68
|
+
answers,
|
|
69
|
+
upstreamModel: payload.model,
|
|
70
|
+
inputTokens,
|
|
71
|
+
outputTokens,
|
|
72
|
+
cost: evaluationCostFor(envelope, spec, { inputTokens, outputTokens }),
|
|
73
|
+
timestamps: { start, first: end, end }
|
|
74
|
+
}
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
function toQuestion (q) {
|
|
78
|
+
if (q.type !== 'binary') return q
|
|
79
|
+
const out = { type: 'noul', instructions: q.instructions }
|
|
80
|
+
if (q.criteria) out.criteria = { true: q.criteria.yes, false: q.criteria.no }
|
|
81
|
+
return out
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
function fromAnswer (id, q, a) {
|
|
85
|
+
if (q.type === 'binary') {
|
|
86
|
+
if (a?.type !== 'noul' || !isProbability(a.noul)) throw mismatch(`no binary answer for '${id}'`)
|
|
87
|
+
return { type: 'binary', probability: a.noul }
|
|
88
|
+
}
|
|
89
|
+
if (a?.type !== q.type || !isProbability(a.confidence)) throw mismatch(`no ${q.type} answer for '${id}'`)
|
|
90
|
+
|
|
91
|
+
if (q.type === 'choice') {
|
|
92
|
+
const options = Object.keys(q.criteria)
|
|
93
|
+
if (typeof a.choice !== 'string' || !options.includes(a.choice) ||
|
|
94
|
+
!options.every(o => isProbability(a.probabilities?.[o]))) {
|
|
95
|
+
throw mismatch(`malformed choice answer for '${id}'`)
|
|
96
|
+
}
|
|
97
|
+
const probabilities = Object.fromEntries(options.map(o => [o, a.probabilities[o]]))
|
|
98
|
+
return { type: 'choice', choice: a.choice, probabilities, confidence: a.confidence }
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
const probabilities = q.criteria.map((_, i) => a.probabilities?.[String(i)])
|
|
102
|
+
if (typeof a.score !== 'number' || !probabilities.every(isProbability)) {
|
|
103
|
+
throw mismatch(`malformed score answer for '${id}'`)
|
|
104
|
+
}
|
|
105
|
+
return { type: 'score', score: a.score, probabilities, confidence: a.confidence }
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
function isProbability (v) {
|
|
109
|
+
return typeof v === 'number' && v >= 0 && v <= 1
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
function isCount (v) {
|
|
113
|
+
return Number.isInteger(v) && v >= 0
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
function mismatch (message) {
|
|
117
|
+
return typedError(message, 'EVALUATE_RESULT_MISMATCH', false)
|
|
118
|
+
}
|
package/js/session/driver.js
CHANGED
|
@@ -19,6 +19,7 @@ import { run } from './run.js'
|
|
|
19
19
|
import { runImage } from './run_image.js'
|
|
20
20
|
import { runTranscription } from './run_transcription.js'
|
|
21
21
|
import { runEmbedding } from './run_embedding.js'
|
|
22
|
+
import { runEvaluation } from './run_evaluation.js'
|
|
22
23
|
import { runInfo } from './run_info.js'
|
|
23
24
|
import { setCatalog } from './adapters/_catalog.js'
|
|
24
25
|
|
|
@@ -233,6 +234,14 @@ export async function drive (stdin, stdout) {
|
|
|
233
234
|
} else {
|
|
234
235
|
await writeLine({ type: 'error', error: out.error })
|
|
235
236
|
}
|
|
237
|
+
} else if (envelope.op === 'evaluate') {
|
|
238
|
+
const { op: _op, ...evEnv } = envelope
|
|
239
|
+
const out = await runEvaluation(evEnv)
|
|
240
|
+
if (out.ok) {
|
|
241
|
+
await writeLine({ type: 'evaluate_done', result: out.result })
|
|
242
|
+
} else {
|
|
243
|
+
await writeLine({ type: 'error', error: out.error })
|
|
244
|
+
}
|
|
236
245
|
} else if (envelope.op === 'info') {
|
|
237
246
|
const out = await runInfo(envelope)
|
|
238
247
|
if (out.ok) {
|
package/js/session/run.js
CHANGED
|
@@ -23,6 +23,7 @@
|
|
|
23
23
|
*/
|
|
24
24
|
|
|
25
25
|
import { ADAPTER_NAMES, isImageProvider } from './adapters/_registry.js'
|
|
26
|
+
import { EVALUATION_PROVIDERS } from './adapters/evaluation/index.js'
|
|
26
27
|
import { getSpec } from './adapters/_catalog.js'
|
|
27
28
|
import { getProviderLimits } from './adapters/_providers.js'
|
|
28
29
|
import { hasSpeed, mergeSpeed, speedHasOwnQuota, speedNames } from './adapters/_speed.js'
|
|
@@ -119,6 +120,13 @@ export async function * run (envelope, {
|
|
|
119
120
|
yield err
|
|
120
121
|
return
|
|
121
122
|
}
|
|
123
|
+
if (EVALUATION_PROVIDERS.includes(provider)) {
|
|
124
|
+
const detail = `provider '${provider}' supports evaluation only; use evaluate(...) instead`
|
|
125
|
+
log.warn({ provider }, '[mohdel:answer] evaluation-only provider via answer')
|
|
126
|
+
endSpanError(span, new Error(detail))
|
|
127
|
+
yield errorEvent(detail, 'PROVIDER_TEXT_NOT_SUPPORTED')
|
|
128
|
+
return
|
|
129
|
+
}
|
|
122
130
|
const err = errorEvent(messageOf(e), 'SESSION_UNKNOWN_PROVIDER')
|
|
123
131
|
log.warn({ err: e, provider }, '[mohdel:answer] unknown provider')
|
|
124
132
|
endSpanError(span, e)
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Evaluation runtime. Resolves the adapter for the envelope's provider and
|
|
3
|
+
* returns either a result or a typed error, never throwing. Rate limits as in
|
|
4
|
+
* `run_embedding.js`, minus the input count.
|
|
5
|
+
*
|
|
6
|
+
* @module session/run_evaluation
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
import { getEvaluationAdapter } from './adapters/evaluation/index.js'
|
|
10
|
+
import { classifyProviderError } from './adapters/_errors.js'
|
|
11
|
+
import { getProviderLimits } from './adapters/_providers.js'
|
|
12
|
+
import * as defaultLimiter from './_rate_limiter.js'
|
|
13
|
+
import { providerOf } from '#core/model-id.js'
|
|
14
|
+
|
|
15
|
+
/**
|
|
16
|
+
* @param {import('#core/evaluation.js').EvaluateEnvelope} envelope
|
|
17
|
+
* @param {{
|
|
18
|
+
* resolveAdapter?: (provider: string) => any,
|
|
19
|
+
* resolveProviderLimits?: (provider: string) => any,
|
|
20
|
+
* limiter?: any,
|
|
21
|
+
* sleep?: (ms: number) => Promise<void>,
|
|
22
|
+
* modelKey?: string,
|
|
23
|
+
* spec?: any
|
|
24
|
+
* }} [options]
|
|
25
|
+
* @returns {Promise<
|
|
26
|
+
* | {ok: true, result: import('#core/evaluation.js').EvaluateResult}
|
|
27
|
+
* | {ok: false, error: import('#core/errors.js').TypedError}
|
|
28
|
+
* >}
|
|
29
|
+
*/
|
|
30
|
+
export async function runEvaluation (envelope, {
|
|
31
|
+
resolveAdapter = getEvaluationAdapter,
|
|
32
|
+
resolveProviderLimits = getProviderLimits,
|
|
33
|
+
limiter = defaultLimiter,
|
|
34
|
+
sleep = defaultSleep,
|
|
35
|
+
modelKey = envelope.model,
|
|
36
|
+
spec
|
|
37
|
+
} = {}) {
|
|
38
|
+
const provider = providerOf(envelope.model)
|
|
39
|
+
|
|
40
|
+
let adapter
|
|
41
|
+
try {
|
|
42
|
+
adapter = resolveAdapter(provider)
|
|
43
|
+
} catch (e) {
|
|
44
|
+
return {
|
|
45
|
+
ok: false,
|
|
46
|
+
error: {
|
|
47
|
+
message: messageOf(e),
|
|
48
|
+
severity: 'error',
|
|
49
|
+
retryable: false,
|
|
50
|
+
type: 'SESSION_UNKNOWN_PROVIDER'
|
|
51
|
+
}
|
|
52
|
+
}
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
const providerCfg = resolveProviderLimits(provider) || {}
|
|
56
|
+
const rpmLimit = spec?.rpmLimit ?? providerCfg.rpmLimit
|
|
57
|
+
const tpmLimit = spec?.tpmLimit ?? providerCfg.tpmLimit
|
|
58
|
+
const bucketKey = spec?.rateLimitScope === 'model' ? modelKey : provider
|
|
59
|
+
|
|
60
|
+
if (rpmLimit != null || tpmLimit != null) {
|
|
61
|
+
const delay = limiter.check(bucketKey, { rpmLimit, tpmLimit })
|
|
62
|
+
if (delay > 0) await sleep(delay)
|
|
63
|
+
limiter.recordRequest(bucketKey)
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
try {
|
|
67
|
+
const result = await adapter(envelope, spec ? { spec } : {})
|
|
68
|
+
if (tpmLimit != null) limiter.recordTokens(bucketKey, result.inputTokens + result.outputTokens)
|
|
69
|
+
return { ok: true, result }
|
|
70
|
+
} catch (e) {
|
|
71
|
+
const typed = /** @type {any} */(e).typed || classifyProviderError(e, envelope.auth?.key)
|
|
72
|
+
return { ok: false, error: typed }
|
|
73
|
+
}
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
/** @param {unknown} e */
|
|
77
|
+
function messageOf (e) {
|
|
78
|
+
return e instanceof Error ? e.message : String(e)
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
/** @param {number} ms */
|
|
82
|
+
function defaultSleep (ms) {
|
|
83
|
+
return new Promise(resolve => setTimeout(resolve, ms))
|
|
84
|
+
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "mohdel",
|
|
3
|
-
"version": "3.
|
|
3
|
+
"version": "3.8.0",
|
|
4
4
|
"license": "MIT",
|
|
5
5
|
"author": {
|
|
6
6
|
"name": "Christophe Le Bars",
|
|
@@ -144,7 +144,7 @@
|
|
|
144
144
|
"@opentelemetry/exporter-trace-otlp-grpc": "^0.222.0",
|
|
145
145
|
"@opentelemetry/sdk-node": "^0.222.0",
|
|
146
146
|
"chalk": "^6.0.1",
|
|
147
|
-
"mohdel-thin-gate-linux-x64-gnu": "3.
|
|
147
|
+
"mohdel-thin-gate-linux-x64-gnu": "3.8.0"
|
|
148
148
|
},
|
|
149
149
|
"dependencies": {
|
|
150
150
|
"@anthropic-ai/sdk": "^0.129.0",
|
package/src/lib/creators.js
CHANGED
|
@@ -65,6 +65,12 @@ const creators = {
|
|
|
65
65
|
logo: 'cohere.svg',
|
|
66
66
|
description: 'Cohere builds retrieval-focused models: embeddings and rerankers aimed at enterprise search rather than chat.'
|
|
67
67
|
},
|
|
68
|
+
typesafe: {
|
|
69
|
+
prefixes: ['jev'],
|
|
70
|
+
label: 'TypeSafe',
|
|
71
|
+
logo: 'typesafe.svg',
|
|
72
|
+
description: 'TypeSafe trains decision models that answer typed questions with calibrated probabilities instead of generating text.'
|
|
73
|
+
},
|
|
68
74
|
nomic: {
|
|
69
75
|
prefixes: ['nomic-embed'],
|
|
70
76
|
label: 'Nomic',
|
package/src/lib/index.js
CHANGED
|
@@ -16,7 +16,7 @@ import {
|
|
|
16
16
|
import { createRateLimiter } from '../../js/session/_rate_limiter.js'
|
|
17
17
|
import { createCooldownTracker } from '../../js/session/_cooldown.js'
|
|
18
18
|
import { setCatalog } from '../../js/session/adapters/_catalog.js'
|
|
19
|
-
import { runAnswer, runAnswerEmbedding, runAnswerImage, runAnswerTranscription } from '../../js/factory/bridge.js'
|
|
19
|
+
import { runAnswer, runAnswerEmbedding, runAnswerEvaluation, runAnswerImage, runAnswerTranscription } from '../../js/factory/bridge.js'
|
|
20
20
|
import { startSpan, endSpanOk, endSpanError } from './tracing.js'
|
|
21
21
|
import { isValidTag } from './schema.js'
|
|
22
22
|
import { silent } from './logger.js'
|
|
@@ -751,6 +751,22 @@ const createModelProxy = (resolvedModelId, modelSpec, handlers, aliasOutputEffor
|
|
|
751
751
|
}
|
|
752
752
|
}
|
|
753
753
|
|
|
754
|
+
if (prop === 'evaluate') {
|
|
755
|
+
return async (state, questions, options = {}) => {
|
|
756
|
+
const { configuration } = await getRuntime()
|
|
757
|
+
return runAnswerEvaluation({
|
|
758
|
+
provider: modelSpec.provider,
|
|
759
|
+
model: modelSpec.model ?? resolvedModelId.split('/').pop(),
|
|
760
|
+
modelKey: resolvedModelId,
|
|
761
|
+
configuration,
|
|
762
|
+
state,
|
|
763
|
+
questions,
|
|
764
|
+
options,
|
|
765
|
+
spec: modelSpec
|
|
766
|
+
}, { limiter: rateLimiter, resolveProviderLimits })
|
|
767
|
+
}
|
|
768
|
+
}
|
|
769
|
+
|
|
754
770
|
if (prop === 'setRateLimit') {
|
|
755
771
|
return async ({ rpm, tpm, inpm } = {}) => {
|
|
756
772
|
const curatedCache = getCuratedCacheSnapshot()
|
package/src/lib/provider-info.js
CHANGED
|
@@ -100,6 +100,13 @@ const PROVIDER_INFO = {
|
|
|
100
100
|
hint: 'Create an API key in the Cohere dashboard under API Keys',
|
|
101
101
|
free: true
|
|
102
102
|
},
|
|
103
|
+
typesafe: {
|
|
104
|
+
label: 'TypeSafe',
|
|
105
|
+
description: 'Calibrated yes/no, choice and score answers about a state. No chat models through mohdel.',
|
|
106
|
+
url: 'https://console.typesafe.ai/keys',
|
|
107
|
+
hint: 'Create an API key in the TypeSafe console under Keys',
|
|
108
|
+
free: false
|
|
109
|
+
},
|
|
103
110
|
qwen: {
|
|
104
111
|
label: 'Qwen Cloud',
|
|
105
112
|
description: 'Qwen — reasoning, coding, long context. Free quota for new users.',
|
package/src/lib/providers.js
CHANGED
|
@@ -144,6 +144,20 @@ const providers = {
|
|
|
144
144
|
rateLimits: 'https://docs.cohere.com/docs/rate-limits'
|
|
145
145
|
}
|
|
146
146
|
},
|
|
147
|
+
typesafe: {
|
|
148
|
+
billing: { kind: 'metered' },
|
|
149
|
+
sdk: 'typesafe',
|
|
150
|
+
api: 'evaluation',
|
|
151
|
+
catalog: false,
|
|
152
|
+
apiKeyEnv: 'TYPESAFE_API_SK',
|
|
153
|
+
baseURL: 'https://api.typesafe.ai/v1',
|
|
154
|
+
createConfiguration: apiKey => ({ apiKey }),
|
|
155
|
+
references: {
|
|
156
|
+
pricing: 'https://docs.typesafe.ai/models',
|
|
157
|
+
models: 'https://docs.typesafe.ai/models',
|
|
158
|
+
rateLimits: 'https://docs.typesafe.ai/models'
|
|
159
|
+
}
|
|
160
|
+
},
|
|
147
161
|
mistral: {
|
|
148
162
|
billing: { kind: 'metered' },
|
|
149
163
|
sdk: 'openai',
|
package/src/lib/schema.js
CHANGED
|
@@ -65,6 +65,7 @@ const fieldDefs = {
|
|
|
65
65
|
maxInputTokens: { type: 'number' },
|
|
66
66
|
inputTypes: { type: 'object' },
|
|
67
67
|
defaultInputType: { type: 'string' },
|
|
68
|
+
evaluationTypes: { type: 'array', itemType: 'string' },
|
|
68
69
|
deprecated: { type: 'string' },
|
|
69
70
|
suspended: { type: 'string' },
|
|
70
71
|
rpmLimit: { type: 'number' },
|