mohdel 3.6.1 → 3.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -0
- package/config/curated.schema.json +8 -2
- package/js/chatgpt/index.js +2 -1
- package/js/client/call_evaluation.js +47 -0
- package/js/client/index.js +1 -0
- package/js/core/evaluation.js +105 -0
- package/js/factory/bridge.js +39 -0
- package/js/session/adapters/_errors.js +4 -2
- package/js/session/adapters/_pricing.js +41 -2
- package/js/session/adapters/embedding/cohere.js +3 -3
- package/js/session/adapters/embedding/gemini.js +3 -3
- package/js/session/adapters/embedding/openai_compatible.js +3 -3
- package/js/session/adapters/evaluation/_shared.js +73 -0
- package/js/session/adapters/evaluation/index.js +22 -0
- package/js/session/adapters/evaluation/typesafe.js +118 -0
- package/js/session/adapters/transcription/openai_compatible.js +2 -2
- package/js/session/driver.js +9 -0
- package/js/session/run.js +8 -0
- package/js/session/run_evaluation.js +84 -0
- package/package.json +2 -2
- package/src/cli/ask.js +7 -3
- package/src/cli/model.js +6 -3
- package/src/lib/catalog-review.js +3 -0
- package/src/lib/creators.js +6 -0
- package/src/lib/index.js +17 -1
- package/src/lib/provider-info.js +7 -0
- package/src/lib/providers.js +49 -0
- package/src/lib/schema.js +1 -0
package/README.md
CHANGED
|
@@ -472,6 +472,7 @@ QWEN_API_SK=sk-...
|
|
|
472
472
|
XIAOMI_API_SK=...
|
|
473
473
|
META_API_SK=...
|
|
474
474
|
COHERE_API_SK=...
|
|
475
|
+
TYPESAFE_API_SK=...
|
|
475
476
|
MOHDEL_LOCAL_API_SK=...
|
|
476
477
|
```
|
|
477
478
|
|
|
@@ -515,6 +516,7 @@ What each provider supports through mohdel's unified interface:
|
|
|
515
516
|
| OpenRouter | Yes | Yes | Yes | No | Varies | Meta-provider; `providerOptions.openrouter` for routing prefs |
|
|
516
517
|
| Local | Yes | Yes | Yes | No | No | Any OpenAI-compatible server; endpoint is the catalog entry's `baseURL` |
|
|
517
518
|
| Cohere | n/a | n/a | n/a | n/a | n/a | Embeddings only: no chat models reach mohdel through it |
|
|
519
|
+
| TypeSafe | n/a | n/a | n/a | n/a | n/a | Evaluation only (`evaluate()`): typed questions answered with probabilities |
|
|
518
520
|
| Novita | Yes | Yes | Yes | No | Yes (`reasoning_content`) | Prices in the model list; text via the shared chat-completions path, separate image adapter |
|
|
519
521
|
|
|
520
522
|
Adapter capability ≠ model capability — whether a given model accepts images, tools, or thinking effort depends on the model spec in `curated.json`. The adapter passes through what the envelope supplies; the provider rejects unsupported combos.
|
|
@@ -87,10 +87,11 @@
|
|
|
87
87
|
"enum": [
|
|
88
88
|
"model",
|
|
89
89
|
"image",
|
|
90
|
-
"transcription"
|
|
90
|
+
"transcription",
|
|
91
|
+
"evaluation"
|
|
91
92
|
],
|
|
92
93
|
"default": "model",
|
|
93
|
-
"description": "'model' for chat/completion, 'image' for image generation, 'transcription' for speech-to-text."
|
|
94
|
+
"description": "'model' for chat/completion, 'image' for image generation, 'transcription' for speech-to-text, 'evaluation' for typed-question answering (evaluate())."
|
|
94
95
|
},
|
|
95
96
|
"label": {
|
|
96
97
|
"type": "string",
|
|
@@ -550,6 +551,11 @@
|
|
|
550
551
|
"type": "string",
|
|
551
552
|
"description": "Symbolic role sent when the caller names none. Required for providers that make the parameter mandatory (Cohere v3+)."
|
|
552
553
|
},
|
|
554
|
+
"evaluationTypes": {
|
|
555
|
+
"type": "array",
|
|
556
|
+
"items": { "type": "string", "enum": ["binary", "choice", "score"] },
|
|
557
|
+
"description": "Question types an evaluation model answers. A question of another type fails before dispatch. Without it, every type is sent and the provider decides."
|
|
558
|
+
},
|
|
553
559
|
"rpmLimit": {
|
|
554
560
|
"type": "integer",
|
|
555
561
|
"minimum": 1,
|
package/js/chatgpt/index.js
CHANGED
|
@@ -4,6 +4,7 @@ import { hostname } from 'node:os'
|
|
|
4
4
|
import { createRemoteJWKSet, jwtVerify } from 'jose'
|
|
5
5
|
import { defaultDirectory, readStore, withStore } from './store.js'
|
|
6
6
|
import { discoverModels, visibleModels } from './models.js'
|
|
7
|
+
import providers from '../../src/lib/providers.js'
|
|
7
8
|
|
|
8
9
|
const ISSUER = 'https://auth.openai.com'
|
|
9
10
|
const RESOURCE = 'https://api.openai.com/v1'
|
|
@@ -15,7 +16,7 @@ const random = () => randomBytes(32).toString('base64url')
|
|
|
15
16
|
const REFRESH_MARGIN = 30000
|
|
16
17
|
// Every registration made before names were stored was sent this hint.
|
|
17
18
|
const UNNAMED_REGISTRATION = 'Mohdel'
|
|
18
|
-
export const usageURL =
|
|
19
|
+
export const usageURL = providers.chatgpt.billing.usage
|
|
19
20
|
|
|
20
21
|
async function jsonRequest (fetcher, url, options = {}) {
|
|
21
22
|
let response
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Send an EvaluateEnvelope to thin-gate's `POST /v1/evaluate`. One-shot: a
|
|
3
|
+
* single JSON response body.
|
|
4
|
+
*
|
|
5
|
+
* @module client/call_evaluation
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
import { requestUnix } from './transport.js'
|
|
9
|
+
import { readAll, parseErrorBody } from './response.js'
|
|
10
|
+
import { MohdelError } from '#core'
|
|
11
|
+
|
|
12
|
+
/**
|
|
13
|
+
* @param {import('#core/evaluation.js').EvaluateEnvelope} envelope
|
|
14
|
+
* @param {object} options
|
|
15
|
+
* @param {string} options.socketPath
|
|
16
|
+
* @param {AbortSignal} [options.signal]
|
|
17
|
+
* @param {string} [options.path] HTTP path; defaults to '/v1/evaluate'
|
|
18
|
+
* @param {Record<string, string>} [options.headers] sent with the request, for a router in front of the gate;
|
|
19
|
+
* `content-type`, `content-length`, `transfer-encoding`, `connection` and `host` are the transport's
|
|
20
|
+
* @returns {Promise<import('#core/evaluation.js').EvaluateResult>}
|
|
21
|
+
*/
|
|
22
|
+
export async function callEvaluation (envelope, { socketPath, signal, path = '/v1/evaluate', headers }) {
|
|
23
|
+
const res = await requestUnix({
|
|
24
|
+
socketPath,
|
|
25
|
+
path,
|
|
26
|
+
method: 'POST',
|
|
27
|
+
body: envelope,
|
|
28
|
+
signal,
|
|
29
|
+
headers
|
|
30
|
+
})
|
|
31
|
+
|
|
32
|
+
const body = await readAll(res)
|
|
33
|
+
|
|
34
|
+
if (res.statusCode !== 200) {
|
|
35
|
+
throw MohdelError.fromJSON(parseErrorBody(body, res.statusCode ?? 0))
|
|
36
|
+
}
|
|
37
|
+
|
|
38
|
+
try {
|
|
39
|
+
return JSON.parse(body)
|
|
40
|
+
} catch (e) {
|
|
41
|
+
throw new MohdelError(`gate returned an unparseable evaluate body: ${body.slice(0, 200)}`, {
|
|
42
|
+
type: 'PROTOCOL_INVALID_RESPONSE',
|
|
43
|
+
severity: 'error',
|
|
44
|
+
retryable: false
|
|
45
|
+
})
|
|
46
|
+
}
|
|
47
|
+
}
|
package/js/client/index.js
CHANGED
|
@@ -18,4 +18,5 @@ export { coalesce } from './coalesce.js'
|
|
|
18
18
|
export { callImage } from './call_image.js'
|
|
19
19
|
export { callTranscription } from './call_transcription.js'
|
|
20
20
|
export { callEmbedding } from './call_embedding.js'
|
|
21
|
+
export { callEvaluation } from './call_evaluation.js'
|
|
21
22
|
export { resolveGateBinary } from './gate-binary.js'
|
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Evaluation envelope and result: one `state`, a map of typed questions, one
|
|
3
|
+
* typed answer per question.
|
|
4
|
+
*
|
|
5
|
+
* @module core/evaluation
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
/**
|
|
9
|
+
* @typedef {object} EvaluateEnvelope
|
|
10
|
+
* @property {string} callId
|
|
11
|
+
* @property {string} authId
|
|
12
|
+
* @property {import('./envelope.js').Auth} auth
|
|
13
|
+
* @property {string} [traceparent]
|
|
14
|
+
* @property {string} [baggage]
|
|
15
|
+
* @property {string} model
|
|
16
|
+
* @property {string | object | any[]} state
|
|
17
|
+
* @property {Record<string, Question>} questions
|
|
18
|
+
* Answers come back under the same ids.
|
|
19
|
+
*/
|
|
20
|
+
|
|
21
|
+
/**
|
|
22
|
+
* @typedef {BinaryQuestion | ChoiceQuestion | ScoreQuestion} Question
|
|
23
|
+
*
|
|
24
|
+
* `instructions` and every criterion description may be a string, an object
|
|
25
|
+
* or an array.
|
|
26
|
+
*/
|
|
27
|
+
|
|
28
|
+
/**
|
|
29
|
+
* @typedef {object} BinaryQuestion
|
|
30
|
+
* @property {'binary'} type
|
|
31
|
+
* @property {any} instructions
|
|
32
|
+
* @property {{yes: any, no: any}} [criteria]
|
|
33
|
+
*/
|
|
34
|
+
|
|
35
|
+
/**
|
|
36
|
+
* @typedef {object} ChoiceQuestion
|
|
37
|
+
* @property {'choice'} type
|
|
38
|
+
* @property {any} instructions
|
|
39
|
+
* @property {Record<string, any>} criteria
|
|
40
|
+
* Option name to its description, or null. At least two options.
|
|
41
|
+
*/
|
|
42
|
+
|
|
43
|
+
/**
|
|
44
|
+
* @typedef {object} ScoreQuestion
|
|
45
|
+
* @property {'score'} type
|
|
46
|
+
* @property {any} instructions
|
|
47
|
+
* @property {any[]} criteria
|
|
48
|
+
* Level descriptions, lowest first; level `i` is worth `i`. At least two.
|
|
49
|
+
*/
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* @typedef {BinaryAnswer | ChoiceAnswer | ScoreAnswer} Answer
|
|
53
|
+
*/
|
|
54
|
+
|
|
55
|
+
/**
|
|
56
|
+
* @typedef {object} BinaryAnswer
|
|
57
|
+
* @property {'binary'} type
|
|
58
|
+
* @property {number} probability
|
|
59
|
+
* Of yes.
|
|
60
|
+
*/
|
|
61
|
+
|
|
62
|
+
/**
|
|
63
|
+
* @typedef {object} ChoiceAnswer
|
|
64
|
+
* @property {'choice'} type
|
|
65
|
+
* @property {string} choice
|
|
66
|
+
* @property {Record<string, number>} probabilities
|
|
67
|
+
* @property {number} [confidence]
|
|
68
|
+
* Absent when the provider reports none.
|
|
69
|
+
*/
|
|
70
|
+
|
|
71
|
+
/**
|
|
72
|
+
* @typedef {object} ScoreAnswer
|
|
73
|
+
* @property {'score'} type
|
|
74
|
+
* @property {number} score
|
|
75
|
+
* Probability-weighted level; can fall between two.
|
|
76
|
+
* @property {number[]} probabilities
|
|
77
|
+
* One per level, in `criteria` order.
|
|
78
|
+
* @property {number} [confidence]
|
|
79
|
+
*/
|
|
80
|
+
|
|
81
|
+
/**
|
|
82
|
+
* @typedef {object} EvaluateResult
|
|
83
|
+
* @property {'completed'} status
|
|
84
|
+
* @property {Record<string, Answer>} answers
|
|
85
|
+
* @property {string} upstreamModel
|
|
86
|
+
* The provider's id for the version that answered, which differs from the
|
|
87
|
+
* entry's `model` when that is an alias.
|
|
88
|
+
* @property {number} inputTokens
|
|
89
|
+
* @property {number} outputTokens
|
|
90
|
+
* @property {number} cost
|
|
91
|
+
* @property {{start: string, first: string, end: string}} timestamps
|
|
92
|
+
*/
|
|
93
|
+
|
|
94
|
+
export const EVALUATE_ENVELOPE_FIELDS = Object.freeze([
|
|
95
|
+
'callId',
|
|
96
|
+
'authId',
|
|
97
|
+
'auth',
|
|
98
|
+
'traceparent',
|
|
99
|
+
'baggage',
|
|
100
|
+
'model',
|
|
101
|
+
'state',
|
|
102
|
+
'questions'
|
|
103
|
+
])
|
|
104
|
+
|
|
105
|
+
export const QUESTION_TYPES = Object.freeze(['binary', 'choice', 'score'])
|
package/js/factory/bridge.js
CHANGED
|
@@ -24,6 +24,7 @@ import { run } from '../session/run.js'
|
|
|
24
24
|
import { runImage } from '../session/run_image.js'
|
|
25
25
|
import { runTranscription } from '../session/run_transcription.js'
|
|
26
26
|
import { runEmbedding } from '../session/run_embedding.js'
|
|
27
|
+
import { runEvaluation } from '../session/run_evaluation.js'
|
|
27
28
|
import { markTrustedMedia } from '../session/adapters/_media.js'
|
|
28
29
|
import { MohdelError, validateIds } from '#core'
|
|
29
30
|
import { createRealtimeDeltaBuffer } from '../../src/lib/utils.js'
|
|
@@ -224,6 +225,44 @@ export async function runAnswerEmbedding ({ provider, model, modelKey, configura
|
|
|
224
225
|
return out.result
|
|
225
226
|
}
|
|
226
227
|
|
|
228
|
+
/**
|
|
229
|
+
* Run an `evaluate()` call through the /session runtime.
|
|
230
|
+
*
|
|
231
|
+
* @param {object} args
|
|
232
|
+
* @param {string} args.provider
|
|
233
|
+
* @param {string} args.model
|
|
234
|
+
* @param {string} [args.modelKey]
|
|
235
|
+
* @param {any} args.configuration
|
|
236
|
+
* @param {import('#core/evaluation.js').EvaluateEnvelope['state']} args.state
|
|
237
|
+
* @param {import('#core/evaluation.js').EvaluateEnvelope['questions']} args.questions
|
|
238
|
+
* @param {any} [args.options] `callId` / `authId` only.
|
|
239
|
+
* @param {any} [args.spec]
|
|
240
|
+
* @param {BridgeDeps} [deps]
|
|
241
|
+
* @returns {Promise<import('#core/evaluation.js').EvaluateResult>}
|
|
242
|
+
*/
|
|
243
|
+
export async function runAnswerEvaluation ({ provider, model, modelKey, configuration, state, questions, options = {}, spec }, deps = {}) {
|
|
244
|
+
const callId = options.callId || newCallId()
|
|
245
|
+
const authId = options.authId || 'local'
|
|
246
|
+
assertValidIds(callId, authId, `${provider}/${model}`)
|
|
247
|
+
|
|
248
|
+
const envelope = {
|
|
249
|
+
callId,
|
|
250
|
+
authId,
|
|
251
|
+
auth: configToAuth(configuration),
|
|
252
|
+
model: `${provider}/${model}`,
|
|
253
|
+
state,
|
|
254
|
+
questions
|
|
255
|
+
}
|
|
256
|
+
|
|
257
|
+
const out = await runEvaluation(envelope, {
|
|
258
|
+
...deps,
|
|
259
|
+
...(modelKey ? { modelKey } : {}),
|
|
260
|
+
...(spec ? { spec } : {})
|
|
261
|
+
})
|
|
262
|
+
if (!out.ok) throw MohdelError.fromJSON(out.error, { provider, model })
|
|
263
|
+
return out.result
|
|
264
|
+
}
|
|
265
|
+
|
|
227
266
|
/**
|
|
228
267
|
* @param {object} args
|
|
229
268
|
* @param {string} args.modelKey Mohdel catalog key `<provider>/<bare>`. The
|
|
@@ -469,9 +469,11 @@ export function typedError (message, type, retryable, detail) {
|
|
|
469
469
|
* @param {number} status
|
|
470
470
|
* @param {string} message
|
|
471
471
|
* @param {string} [detail]
|
|
472
|
+
* @param {string} [key] masked out of `detail`
|
|
472
473
|
* @returns {Error & {typed: import('#core/errors.js').TypedError}}
|
|
473
474
|
*/
|
|
474
|
-
export function fromHttpStatus (status, message, detail) {
|
|
475
|
+
export function fromHttpStatus (status, message, detail, key) {
|
|
475
476
|
const typed = classifyProviderError({ status })
|
|
476
|
-
|
|
477
|
+
const scrubbed = scrubKey(detail, key)
|
|
478
|
+
return typedError(typed.message, typed.type, typed.retryable, scrubbed ? `${message}: ${scrubbed}` : message)
|
|
477
479
|
}
|
|
@@ -10,6 +10,16 @@
|
|
|
10
10
|
*/
|
|
11
11
|
|
|
12
12
|
import { setCatalog, specFor } from './_catalog.js'
|
|
13
|
+
import { providerOf } from '#core/model-id.js'
|
|
14
|
+
import providers from '../../../src/lib/providers.js'
|
|
15
|
+
|
|
16
|
+
// `plan` and `capacity` calls have no per-call API invoice, so `cost`, which is
|
|
17
|
+
// API USD, is 0 whatever the catalog prices. `fake` and `echo` are adapters,
|
|
18
|
+
// not providers: they carry no declaration and price from the catalog.
|
|
19
|
+
function unmetered (model) {
|
|
20
|
+
const def = providers[providerOf(model)]
|
|
21
|
+
return def !== undefined && def.billing.kind !== 'metered'
|
|
22
|
+
}
|
|
13
23
|
|
|
14
24
|
/**
|
|
15
25
|
* Pure cost computation from spec + usage.
|
|
@@ -112,11 +122,40 @@ function resolveTier (price, tokens) {
|
|
|
112
122
|
* @returns {number}
|
|
113
123
|
*/
|
|
114
124
|
export function costFor (envelope, usage) {
|
|
115
|
-
|
|
116
|
-
if (envelope.model.startsWith('chatgpt/')) return 0
|
|
125
|
+
if (unmetered(envelope.model)) return 0
|
|
117
126
|
return computeCost(specFor(envelope), usage)
|
|
118
127
|
}
|
|
119
128
|
|
|
129
|
+
/**
|
|
130
|
+
* @param {{model: string}} envelope
|
|
131
|
+
* @param {any} spec
|
|
132
|
+
* @param {{durationSeconds?: number | null, inputTokens?: number, outputTokens?: number}} usage
|
|
133
|
+
* @returns {number}
|
|
134
|
+
*/
|
|
135
|
+
export function transcriptionCostFor (envelope, spec, usage) {
|
|
136
|
+
return unmetered(envelope.model) ? 0 : computeTranscriptionCost(spec, usage)
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
/**
|
|
140
|
+
* @param {{model: string}} envelope
|
|
141
|
+
* @param {any} spec
|
|
142
|
+
* @param {{inputTokens?: number}} usage
|
|
143
|
+
* @returns {number}
|
|
144
|
+
*/
|
|
145
|
+
export function embeddingCostFor (envelope, spec, usage) {
|
|
146
|
+
return unmetered(envelope.model) ? 0 : computeEmbeddingCost(spec, usage)
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
/**
|
|
150
|
+
* @param {{model: string}} envelope
|
|
151
|
+
* @param {any} spec
|
|
152
|
+
* @param {{inputTokens: number, outputTokens: number}} usage
|
|
153
|
+
* @returns {number}
|
|
154
|
+
*/
|
|
155
|
+
export function evaluationCostFor (envelope, spec, usage) {
|
|
156
|
+
return unmetered(envelope.model) ? 0 : computeCost(spec, usage)
|
|
157
|
+
}
|
|
158
|
+
|
|
120
159
|
/**
|
|
121
160
|
* Cost of a transcription call.
|
|
122
161
|
*
|
|
@@ -12,7 +12,7 @@
|
|
|
12
12
|
|
|
13
13
|
import { getSpec } from '../_catalog.js'
|
|
14
14
|
import { classifyProviderError, fromHttpStatus, typedError } from '../_errors.js'
|
|
15
|
-
import {
|
|
15
|
+
import { embeddingCostFor } from '../_pricing.js'
|
|
16
16
|
import { catalogKey, bareOf } from '#core/model-id.js'
|
|
17
17
|
import { checkBatch, checkDimensions, resolveInputType, widthOf } from './_shared.js'
|
|
18
18
|
|
|
@@ -61,7 +61,7 @@ export async function cohereEmbedding (envelope, deps = {}) {
|
|
|
61
61
|
|
|
62
62
|
if (!res.ok) {
|
|
63
63
|
const detail = await res.text().catch(() => '')
|
|
64
|
-
throw fromHttpStatus(res.status, detail, envelope.auth?.key)
|
|
64
|
+
throw fromHttpStatus(res.status, 'embedding request failed', detail.slice(0, 500), envelope.auth?.key)
|
|
65
65
|
}
|
|
66
66
|
|
|
67
67
|
const payload = await res.json()
|
|
@@ -85,7 +85,7 @@ export async function cohereEmbedding (envelope, deps = {}) {
|
|
|
85
85
|
dimensions: widthOf(vectors),
|
|
86
86
|
inputType,
|
|
87
87
|
inputTokens,
|
|
88
|
-
cost:
|
|
88
|
+
cost: embeddingCostFor(envelope, spec, { inputTokens }),
|
|
89
89
|
timestamps: { start, first: end, end }
|
|
90
90
|
}
|
|
91
91
|
}
|
|
@@ -13,7 +13,7 @@
|
|
|
13
13
|
|
|
14
14
|
import { getSpec } from '../_catalog.js'
|
|
15
15
|
import { classifyProviderError, fromHttpStatus, typedError } from '../_errors.js'
|
|
16
|
-
import {
|
|
16
|
+
import { embeddingCostFor } from '../_pricing.js'
|
|
17
17
|
import { catalogKey, bareOf } from '#core/model-id.js'
|
|
18
18
|
import { checkBatch, checkDimensions, resolveInputType, widthOf } from './_shared.js'
|
|
19
19
|
|
|
@@ -59,7 +59,7 @@ export async function geminiEmbedding (envelope, deps = {}) {
|
|
|
59
59
|
|
|
60
60
|
if (!res.ok) {
|
|
61
61
|
const detail = await res.text().catch(() => '')
|
|
62
|
-
throw fromHttpStatus(res.status, detail, envelope.auth?.key)
|
|
62
|
+
throw fromHttpStatus(res.status, 'embedding request failed', detail.slice(0, 500), envelope.auth?.key)
|
|
63
63
|
}
|
|
64
64
|
|
|
65
65
|
const payload = await res.json()
|
|
@@ -86,7 +86,7 @@ export async function geminiEmbedding (envelope, deps = {}) {
|
|
|
86
86
|
dimensions: widthOf(vectors),
|
|
87
87
|
inputType: taskType,
|
|
88
88
|
inputTokens,
|
|
89
|
-
cost:
|
|
89
|
+
cost: embeddingCostFor(envelope, spec, { inputTokens }),
|
|
90
90
|
timestamps: { start, first: end, end }
|
|
91
91
|
}
|
|
92
92
|
}
|
|
@@ -11,7 +11,7 @@
|
|
|
11
11
|
|
|
12
12
|
import { getSpec } from '../_catalog.js'
|
|
13
13
|
import { classifyProviderError, fromHttpStatus, typedError } from '../_errors.js'
|
|
14
|
-
import {
|
|
14
|
+
import { embeddingCostFor } from '../_pricing.js'
|
|
15
15
|
import { catalogKey, bareOf } from '#core/model-id.js'
|
|
16
16
|
import { checkBatch, checkDimensions, resolveInputType, widthOf } from './_shared.js'
|
|
17
17
|
|
|
@@ -56,7 +56,7 @@ export function createEmbeddingAdapter ({ baseURL, dimensionsField = 'dimensions
|
|
|
56
56
|
|
|
57
57
|
if (!res.ok) {
|
|
58
58
|
const detail = await res.text().catch(() => '')
|
|
59
|
-
throw fromHttpStatus(res.status, detail, envelope.auth?.key)
|
|
59
|
+
throw fromHttpStatus(res.status, 'embedding request failed', detail.slice(0, 500), envelope.auth?.key)
|
|
60
60
|
}
|
|
61
61
|
|
|
62
62
|
const payload = await res.json()
|
|
@@ -85,7 +85,7 @@ export function createEmbeddingAdapter ({ baseURL, dimensionsField = 'dimensions
|
|
|
85
85
|
dimensions: widthOf(vectors),
|
|
86
86
|
inputType,
|
|
87
87
|
inputTokens,
|
|
88
|
-
cost:
|
|
88
|
+
cost: embeddingCostFor(envelope, spec, { inputTokens }),
|
|
89
89
|
timestamps: { start, first: end, end }
|
|
90
90
|
}
|
|
91
91
|
}
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Pre-dispatch checks every evaluation adapter runs. They reject what the
|
|
3
|
+
* gate's serde types reject, so a malformed envelope fails the same way
|
|
4
|
+
* in-process.
|
|
5
|
+
*
|
|
6
|
+
* @module session/adapters/evaluation/shared
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
import { typedError } from '../_errors.js'
|
|
10
|
+
import { QUESTION_TYPES } from '#core/evaluation.js'
|
|
11
|
+
|
|
12
|
+
const QUESTION_KEYS = new Set(['type', 'instructions', 'criteria'])
|
|
13
|
+
const BINARY_CRITERIA_KEYS = ['yes', 'no']
|
|
14
|
+
|
|
15
|
+
/**
|
|
16
|
+
* @param {import('#core/evaluation.js').EvaluateEnvelope} envelope
|
|
17
|
+
* @param {any} spec
|
|
18
|
+
*/
|
|
19
|
+
export function checkEvaluation (envelope, spec) {
|
|
20
|
+
const { state, questions } = envelope
|
|
21
|
+
if (typeof state !== 'string' && !isObject(state) && !Array.isArray(state)) {
|
|
22
|
+
throw invalid('state must be a string, an object or an array')
|
|
23
|
+
}
|
|
24
|
+
if (!isObject(questions) || Object.keys(questions).length === 0) {
|
|
25
|
+
throw invalid('questions must be a non-empty object')
|
|
26
|
+
}
|
|
27
|
+
for (const [id, q] of Object.entries(questions)) {
|
|
28
|
+
checkQuestion(id, q, spec)
|
|
29
|
+
}
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
function checkQuestion (id, q, spec) {
|
|
33
|
+
if (!isObject(q)) throw invalid(`question '${id}' must be an object`)
|
|
34
|
+
const extra = Object.keys(q).filter(k => !QUESTION_KEYS.has(k))
|
|
35
|
+
if (extra.length) throw invalid(`question '${id}' has unknown field '${extra.join("', '")}'`)
|
|
36
|
+
if (!QUESTION_TYPES.includes(q.type)) {
|
|
37
|
+
throw invalid(`question '${id}' type must be one of ${QUESTION_TYPES.join(', ')}`)
|
|
38
|
+
}
|
|
39
|
+
if (Array.isArray(spec?.evaluationTypes) && !spec.evaluationTypes.includes(q.type)) {
|
|
40
|
+
throw typedError(
|
|
41
|
+
`question '${id}' is '${q.type}', which this model does not answer; it answers ${spec.evaluationTypes.join(', ')}`,
|
|
42
|
+
'EVALUATE_QUESTION_TYPE_UNSUPPORTED',
|
|
43
|
+
false
|
|
44
|
+
)
|
|
45
|
+
}
|
|
46
|
+
if (q.instructions === undefined || q.instructions === null) {
|
|
47
|
+
throw invalid(`question '${id}' requires instructions`)
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
const c = q.criteria
|
|
51
|
+
if (q.type === 'binary') {
|
|
52
|
+
if (c === undefined) return
|
|
53
|
+
if (!isObject(c) || Object.keys(c).length !== 2 || !BINARY_CRITERIA_KEYS.every(k => c[k] !== undefined && c[k] !== null)) {
|
|
54
|
+
throw invalid(`question '${id}' criteria must have exactly 'yes' and 'no'`)
|
|
55
|
+
}
|
|
56
|
+
} else if (q.type === 'choice') {
|
|
57
|
+
if (!isObject(c) || Object.keys(c).length < 2) {
|
|
58
|
+
throw invalid(`question '${id}' criteria must name at least two options`)
|
|
59
|
+
}
|
|
60
|
+
} else if (!Array.isArray(c) || c.length < 2) {
|
|
61
|
+
throw invalid(`question '${id}' criteria must list at least two levels`)
|
|
62
|
+
}
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
/** @param {unknown} v */
|
|
66
|
+
function isObject (v) {
|
|
67
|
+
return typeof v === 'object' && v !== null && !Array.isArray(v)
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
/** @param {string} message */
|
|
71
|
+
function invalid (message) {
|
|
72
|
+
return typedError(message, 'EVALUATE_INPUT_INVALID', false)
|
|
73
|
+
}
|
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Evaluation-adapter registry.
|
|
3
|
+
*
|
|
4
|
+
* @module session/adapters/evaluation
|
|
5
|
+
*/
|
|
6
|
+
|
|
7
|
+
import { typesafeEvaluation } from './typesafe.js'
|
|
8
|
+
|
|
9
|
+
const EVALUATION_ADAPTERS = {
|
|
10
|
+
typesafe: typesafeEvaluation
|
|
11
|
+
}
|
|
12
|
+
|
|
13
|
+
export const EVALUATION_PROVIDERS = Object.freeze(Object.keys(EVALUATION_ADAPTERS))
|
|
14
|
+
|
|
15
|
+
/**
|
|
16
|
+
* @param {string} provider
|
|
17
|
+
*/
|
|
18
|
+
export function getEvaluationAdapter (provider) {
|
|
19
|
+
const adapter = EVALUATION_ADAPTERS[provider]
|
|
20
|
+
if (!adapter) throw new Error(`no evaluation adapter for provider: ${provider}`)
|
|
21
|
+
return adapter
|
|
22
|
+
}
|
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* TypeSafe evaluation adapter (`POST /v1/systemone`). TypeSafe calls a binary
|
|
3
|
+
* question `noul` and keys binary criteria `true` / `false`; score
|
|
4
|
+
* probabilities come back keyed by level index as strings.
|
|
5
|
+
*
|
|
6
|
+
* @module session/adapters/evaluation/typesafe
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
import { getSpec } from '../_catalog.js'
|
|
10
|
+
import { classifyProviderError, fromHttpStatus, typedError } from '../_errors.js'
|
|
11
|
+
import { evaluationCostFor } from '../_pricing.js'
|
|
12
|
+
import { catalogKey, bareOf } from '#core/model-id.js'
|
|
13
|
+
import { checkEvaluation } from './_shared.js'
|
|
14
|
+
|
|
15
|
+
const BASE_URL = 'https://api.typesafe.ai/v1'
|
|
16
|
+
|
|
17
|
+
export async function typesafeEvaluation (envelope, deps = {}) {
|
|
18
|
+
const fetchFn = deps.fetch ?? globalThis.fetch
|
|
19
|
+
const spec = deps.spec ?? getSpec(catalogKey(envelope.model)) ?? {}
|
|
20
|
+
const start = String(process.hrtime.bigint())
|
|
21
|
+
|
|
22
|
+
checkEvaluation(envelope, spec)
|
|
23
|
+
|
|
24
|
+
const body = {
|
|
25
|
+
model: spec.model ?? bareOf(envelope.model),
|
|
26
|
+
state: envelope.state,
|
|
27
|
+
questions: Object.fromEntries(
|
|
28
|
+
Object.entries(envelope.questions).map(([id, q]) => [id, toQuestion(q)])
|
|
29
|
+
)
|
|
30
|
+
}
|
|
31
|
+
|
|
32
|
+
let res
|
|
33
|
+
try {
|
|
34
|
+
res = await fetchFn(`${BASE_URL}/systemone`, {
|
|
35
|
+
method: 'POST',
|
|
36
|
+
headers: {
|
|
37
|
+
'Content-Type': 'application/json',
|
|
38
|
+
Authorization: `Bearer ${envelope.auth.key}`
|
|
39
|
+
},
|
|
40
|
+
body: JSON.stringify(body)
|
|
41
|
+
})
|
|
42
|
+
} catch (e) {
|
|
43
|
+
throw typedError(classifyProviderError(e, envelope.auth?.key).message, 'NET_ERROR', true)
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
if (!res.ok) {
|
|
47
|
+
const detail = await res.text().catch(() => '')
|
|
48
|
+
throw fromHttpStatus(res.status, 'evaluation request failed', detail.slice(0, 500), envelope.auth?.key)
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
const payload = await res.json()
|
|
52
|
+
const usage = payload?.usage
|
|
53
|
+
if (typeof payload?.model !== 'string' || !isCount(usage?.input_tokens) || !isCount(usage?.output_tokens)) {
|
|
54
|
+
throw mismatch('response lacks model or token usage')
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
const answers = {}
|
|
58
|
+
for (const [id, q] of Object.entries(envelope.questions)) {
|
|
59
|
+
answers[id] = fromAnswer(id, q, payload.answers?.[id])
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
const inputTokens = usage.input_tokens
|
|
63
|
+
const outputTokens = usage.output_tokens
|
|
64
|
+
const end = String(process.hrtime.bigint())
|
|
65
|
+
|
|
66
|
+
return {
|
|
67
|
+
status: 'completed',
|
|
68
|
+
answers,
|
|
69
|
+
upstreamModel: payload.model,
|
|
70
|
+
inputTokens,
|
|
71
|
+
outputTokens,
|
|
72
|
+
cost: evaluationCostFor(envelope, spec, { inputTokens, outputTokens }),
|
|
73
|
+
timestamps: { start, first: end, end }
|
|
74
|
+
}
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
function toQuestion (q) {
|
|
78
|
+
if (q.type !== 'binary') return q
|
|
79
|
+
const out = { type: 'noul', instructions: q.instructions }
|
|
80
|
+
if (q.criteria) out.criteria = { true: q.criteria.yes, false: q.criteria.no }
|
|
81
|
+
return out
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
function fromAnswer (id, q, a) {
|
|
85
|
+
if (q.type === 'binary') {
|
|
86
|
+
if (a?.type !== 'noul' || !isProbability(a.noul)) throw mismatch(`no binary answer for '${id}'`)
|
|
87
|
+
return { type: 'binary', probability: a.noul }
|
|
88
|
+
}
|
|
89
|
+
if (a?.type !== q.type || !isProbability(a.confidence)) throw mismatch(`no ${q.type} answer for '${id}'`)
|
|
90
|
+
|
|
91
|
+
if (q.type === 'choice') {
|
|
92
|
+
const options = Object.keys(q.criteria)
|
|
93
|
+
if (typeof a.choice !== 'string' || !options.includes(a.choice) ||
|
|
94
|
+
!options.every(o => isProbability(a.probabilities?.[o]))) {
|
|
95
|
+
throw mismatch(`malformed choice answer for '${id}'`)
|
|
96
|
+
}
|
|
97
|
+
const probabilities = Object.fromEntries(options.map(o => [o, a.probabilities[o]]))
|
|
98
|
+
return { type: 'choice', choice: a.choice, probabilities, confidence: a.confidence }
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
const probabilities = q.criteria.map((_, i) => a.probabilities?.[String(i)])
|
|
102
|
+
if (typeof a.score !== 'number' || !probabilities.every(isProbability)) {
|
|
103
|
+
throw mismatch(`malformed score answer for '${id}'`)
|
|
104
|
+
}
|
|
105
|
+
return { type: 'score', score: a.score, probabilities, confidence: a.confidence }
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
function isProbability (v) {
|
|
109
|
+
return typeof v === 'number' && v >= 0 && v <= 1
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
function isCount (v) {
|
|
113
|
+
return Number.isInteger(v) && v >= 0
|
|
114
|
+
}
|
|
115
|
+
|
|
116
|
+
function mismatch (message) {
|
|
117
|
+
return typedError(message, 'EVALUATE_RESULT_MISMATCH', false)
|
|
118
|
+
}
|
|
@@ -23,7 +23,7 @@ import { basename } from 'node:path'
|
|
|
23
23
|
import { getSpec } from '../_catalog.js'
|
|
24
24
|
import { classifyProviderError, fromHttpStatus, typedError } from '../_errors.js'
|
|
25
25
|
import { dataUriPayload, isTrustedMedia, mediaScheme, readLocalMedia } from '../_media.js'
|
|
26
|
-
import {
|
|
26
|
+
import { transcriptionCostFor } from '../_pricing.js'
|
|
27
27
|
import { catalogKey, bareOf } from '#core/model-id.js'
|
|
28
28
|
|
|
29
29
|
/**
|
|
@@ -67,7 +67,7 @@ export function createTranscriptionAdapter ({ baseURL, responseFormat }) {
|
|
|
67
67
|
const body = await res.json()
|
|
68
68
|
const durationSeconds = extractDuration(body)
|
|
69
69
|
const tokens = extractTokens(body)
|
|
70
|
-
const cost =
|
|
70
|
+
const cost = transcriptionCostFor(envelope, spec, { durationSeconds, ...tokens })
|
|
71
71
|
|
|
72
72
|
const end = String(process.hrtime.bigint())
|
|
73
73
|
return {
|
package/js/session/driver.js
CHANGED
|
@@ -19,6 +19,7 @@ import { run } from './run.js'
|
|
|
19
19
|
import { runImage } from './run_image.js'
|
|
20
20
|
import { runTranscription } from './run_transcription.js'
|
|
21
21
|
import { runEmbedding } from './run_embedding.js'
|
|
22
|
+
import { runEvaluation } from './run_evaluation.js'
|
|
22
23
|
import { runInfo } from './run_info.js'
|
|
23
24
|
import { setCatalog } from './adapters/_catalog.js'
|
|
24
25
|
|
|
@@ -233,6 +234,14 @@ export async function drive (stdin, stdout) {
|
|
|
233
234
|
} else {
|
|
234
235
|
await writeLine({ type: 'error', error: out.error })
|
|
235
236
|
}
|
|
237
|
+
} else if (envelope.op === 'evaluate') {
|
|
238
|
+
const { op: _op, ...evEnv } = envelope
|
|
239
|
+
const out = await runEvaluation(evEnv)
|
|
240
|
+
if (out.ok) {
|
|
241
|
+
await writeLine({ type: 'evaluate_done', result: out.result })
|
|
242
|
+
} else {
|
|
243
|
+
await writeLine({ type: 'error', error: out.error })
|
|
244
|
+
}
|
|
236
245
|
} else if (envelope.op === 'info') {
|
|
237
246
|
const out = await runInfo(envelope)
|
|
238
247
|
if (out.ok) {
|
package/js/session/run.js
CHANGED
|
@@ -23,6 +23,7 @@
|
|
|
23
23
|
*/
|
|
24
24
|
|
|
25
25
|
import { ADAPTER_NAMES, isImageProvider } from './adapters/_registry.js'
|
|
26
|
+
import { EVALUATION_PROVIDERS } from './adapters/evaluation/index.js'
|
|
26
27
|
import { getSpec } from './adapters/_catalog.js'
|
|
27
28
|
import { getProviderLimits } from './adapters/_providers.js'
|
|
28
29
|
import { hasSpeed, mergeSpeed, speedHasOwnQuota, speedNames } from './adapters/_speed.js'
|
|
@@ -119,6 +120,13 @@ export async function * run (envelope, {
|
|
|
119
120
|
yield err
|
|
120
121
|
return
|
|
121
122
|
}
|
|
123
|
+
if (EVALUATION_PROVIDERS.includes(provider)) {
|
|
124
|
+
const detail = `provider '${provider}' supports evaluation only; use evaluate(...) instead`
|
|
125
|
+
log.warn({ provider }, '[mohdel:answer] evaluation-only provider via answer')
|
|
126
|
+
endSpanError(span, new Error(detail))
|
|
127
|
+
yield errorEvent(detail, 'PROVIDER_TEXT_NOT_SUPPORTED')
|
|
128
|
+
return
|
|
129
|
+
}
|
|
122
130
|
const err = errorEvent(messageOf(e), 'SESSION_UNKNOWN_PROVIDER')
|
|
123
131
|
log.warn({ err: e, provider }, '[mohdel:answer] unknown provider')
|
|
124
132
|
endSpanError(span, e)
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Evaluation runtime. Resolves the adapter for the envelope's provider and
|
|
3
|
+
* returns either a result or a typed error, never throwing. Rate limits as in
|
|
4
|
+
* `run_embedding.js`, minus the input count.
|
|
5
|
+
*
|
|
6
|
+
* @module session/run_evaluation
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
import { getEvaluationAdapter } from './adapters/evaluation/index.js'
|
|
10
|
+
import { classifyProviderError } from './adapters/_errors.js'
|
|
11
|
+
import { getProviderLimits } from './adapters/_providers.js'
|
|
12
|
+
import * as defaultLimiter from './_rate_limiter.js'
|
|
13
|
+
import { providerOf } from '#core/model-id.js'
|
|
14
|
+
|
|
15
|
+
/**
|
|
16
|
+
* @param {import('#core/evaluation.js').EvaluateEnvelope} envelope
|
|
17
|
+
* @param {{
|
|
18
|
+
* resolveAdapter?: (provider: string) => any,
|
|
19
|
+
* resolveProviderLimits?: (provider: string) => any,
|
|
20
|
+
* limiter?: any,
|
|
21
|
+
* sleep?: (ms: number) => Promise<void>,
|
|
22
|
+
* modelKey?: string,
|
|
23
|
+
* spec?: any
|
|
24
|
+
* }} [options]
|
|
25
|
+
* @returns {Promise<
|
|
26
|
+
* | {ok: true, result: import('#core/evaluation.js').EvaluateResult}
|
|
27
|
+
* | {ok: false, error: import('#core/errors.js').TypedError}
|
|
28
|
+
* >}
|
|
29
|
+
*/
|
|
30
|
+
export async function runEvaluation (envelope, {
|
|
31
|
+
resolveAdapter = getEvaluationAdapter,
|
|
32
|
+
resolveProviderLimits = getProviderLimits,
|
|
33
|
+
limiter = defaultLimiter,
|
|
34
|
+
sleep = defaultSleep,
|
|
35
|
+
modelKey = envelope.model,
|
|
36
|
+
spec
|
|
37
|
+
} = {}) {
|
|
38
|
+
const provider = providerOf(envelope.model)
|
|
39
|
+
|
|
40
|
+
let adapter
|
|
41
|
+
try {
|
|
42
|
+
adapter = resolveAdapter(provider)
|
|
43
|
+
} catch (e) {
|
|
44
|
+
return {
|
|
45
|
+
ok: false,
|
|
46
|
+
error: {
|
|
47
|
+
message: messageOf(e),
|
|
48
|
+
severity: 'error',
|
|
49
|
+
retryable: false,
|
|
50
|
+
type: 'SESSION_UNKNOWN_PROVIDER'
|
|
51
|
+
}
|
|
52
|
+
}
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
const providerCfg = resolveProviderLimits(provider) || {}
|
|
56
|
+
const rpmLimit = spec?.rpmLimit ?? providerCfg.rpmLimit
|
|
57
|
+
const tpmLimit = spec?.tpmLimit ?? providerCfg.tpmLimit
|
|
58
|
+
const bucketKey = spec?.rateLimitScope === 'model' ? modelKey : provider
|
|
59
|
+
|
|
60
|
+
if (rpmLimit != null || tpmLimit != null) {
|
|
61
|
+
const delay = limiter.check(bucketKey, { rpmLimit, tpmLimit })
|
|
62
|
+
if (delay > 0) await sleep(delay)
|
|
63
|
+
limiter.recordRequest(bucketKey)
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
try {
|
|
67
|
+
const result = await adapter(envelope, spec ? { spec } : {})
|
|
68
|
+
if (tpmLimit != null) limiter.recordTokens(bucketKey, result.inputTokens + result.outputTokens)
|
|
69
|
+
return { ok: true, result }
|
|
70
|
+
} catch (e) {
|
|
71
|
+
const typed = /** @type {any} */(e).typed || classifyProviderError(e, envelope.auth?.key)
|
|
72
|
+
return { ok: false, error: typed }
|
|
73
|
+
}
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
/** @param {unknown} e */
|
|
77
|
+
function messageOf (e) {
|
|
78
|
+
return e instanceof Error ? e.message : String(e)
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
/** @param {number} ms */
|
|
82
|
+
function defaultSleep (ms) {
|
|
83
|
+
return new Promise(resolve => setTimeout(resolve, ms))
|
|
84
|
+
}
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "mohdel",
|
|
3
|
-
"version": "3.
|
|
3
|
+
"version": "3.8.0",
|
|
4
4
|
"license": "MIT",
|
|
5
5
|
"author": {
|
|
6
6
|
"name": "Christophe Le Bars",
|
|
@@ -144,7 +144,7 @@
|
|
|
144
144
|
"@opentelemetry/exporter-trace-otlp-grpc": "^0.222.0",
|
|
145
145
|
"@opentelemetry/sdk-node": "^0.222.0",
|
|
146
146
|
"chalk": "^6.0.1",
|
|
147
|
-
"mohdel-thin-gate-linux-x64-gnu": "3.
|
|
147
|
+
"mohdel-thin-gate-linux-x64-gnu": "3.8.0"
|
|
148
148
|
},
|
|
149
149
|
"dependencies": {
|
|
150
150
|
"@anthropic-ai/sdk": "^0.129.0",
|
package/src/cli/ask.js
CHANGED
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
import mohdel, { silent } from '../lib/index.js'
|
|
2
2
|
import { getConfig, loadDefaultEnv } from '../lib/common.js'
|
|
3
|
+
import providerDefs, { billingOf } from '../lib/providers.js'
|
|
3
4
|
|
|
4
5
|
const noop = () => {}
|
|
5
6
|
|
|
@@ -47,8 +48,10 @@ export const hintsForError = (err, modelId) => {
|
|
|
47
48
|
else hints.push('→ run: mo # interactive provider/key setup')
|
|
48
49
|
}
|
|
49
50
|
|
|
50
|
-
|
|
51
|
-
|
|
51
|
+
// The id is what the user typed, so its provider may not exist.
|
|
52
|
+
const usage = providerDefs[provider]?.billing.usage
|
|
53
|
+
if (usage && /RATE_LIMIT|QUOTA_EXHAUSTED|429|usage limit/i.test(`${err?.type || ''} ${both}`)) {
|
|
54
|
+
hints.push(`→ manage ${providerDefs[provider].billing.label} usage: ${usage}`)
|
|
52
55
|
}
|
|
53
56
|
|
|
54
57
|
if (/deprecated/i.test(both) && /replacement/i.test(both)) {
|
|
@@ -223,7 +226,8 @@ Examples:
|
|
|
223
226
|
if (tokens.inputTokens) summary.push(`${tokens.inputTokens} in`)
|
|
224
227
|
if (tokens.outputTokens) summary.push(`${tokens.outputTokens} out`)
|
|
225
228
|
if (tokens.thinkingTokens) summary.push(`${tokens.thinkingTokens} think`)
|
|
226
|
-
|
|
229
|
+
const billing = billingOf(model.id)
|
|
230
|
+
if (billing.kind !== 'metered') summary.push(`not metered · ${billing.label}${billing.usage ? ` — manage usage: ${billing.usage}` : ''}`)
|
|
227
231
|
else if (tokens.cost != null) summary.push(`$${tokens.cost.toFixed(4)}`)
|
|
228
232
|
if (tokens.speed) {
|
|
229
233
|
const served = tokens.servedSpeed
|
package/src/cli/model.js
CHANGED
|
@@ -731,11 +731,11 @@ Alibaba's Qwen. Routing follows the provider — see "mo provider --help".`)
|
|
|
731
731
|
|
|
732
732
|
// The provider API returns ids, never prices — curated entries land unpriced.
|
|
733
733
|
function printCurateNext (providerName) {
|
|
734
|
-
|
|
735
|
-
|
|
734
|
+
const def = providerDefs[providerName]
|
|
735
|
+
if (def.billing.kind !== 'metered') {
|
|
736
|
+
console.log(`${providerName} models are ready. Use mo ask ${providerName}/<model-id> "your prompt". Calls are not metered: they draw on your ${def.billing.label}.`)
|
|
736
737
|
return
|
|
737
738
|
}
|
|
738
|
-
const def = providerDefs[providerName]
|
|
739
739
|
if (def?.pricesFromApi) {
|
|
740
740
|
console.log(`\n${meta(`${providerName} publishes prices in its model list — the entries are complete.`)}`)
|
|
741
741
|
console.log(`${meta('Check them:')} mo ls ${meta('│')} mo check`)
|
|
@@ -779,6 +779,9 @@ const SINGLE_DIMENSION_PRICES = [
|
|
|
779
779
|
]
|
|
780
780
|
|
|
781
781
|
function formatPrice (info) {
|
|
782
|
+
// An entry naming an unknown provider is `mo check`'s to report; it lists with its prices.
|
|
783
|
+
const billing = providerDefs[info.provider]?.billing
|
|
784
|
+
if (billing && billing.kind !== 'metered') return meta(`not metered · ${billing.label}`)
|
|
782
785
|
const inp = resolvePrice(info.inputPrice)
|
|
783
786
|
const out = resolvePrice(info.outputPrice)
|
|
784
787
|
if (inp || out) return price(`$${inp}`) + meta('/') + price(`$${out}`)
|
|
@@ -73,6 +73,9 @@ export const reviewEntry = (key, spec, catalog, { strict = false, local = null }
|
|
|
73
73
|
if (spec.provider && spec.provider !== keyProvider) {
|
|
74
74
|
errors.push(`${key}: spec.provider '${spec.provider}' doesn't match key prefix '${keyProvider}'`)
|
|
75
75
|
}
|
|
76
|
+
if (providerConfig?.billing.kind === 'capacity' && Object.keys(spec).some(k => k.endsWith('Price'))) {
|
|
77
|
+
warnings.push(`${key}: prices are never charged — ${providerConfig.billing.label} calls report cost 0`)
|
|
78
|
+
}
|
|
76
79
|
if (providerConfig && spec.sdk && spec.sdk !== providerConfig.sdk) {
|
|
77
80
|
errors.push(`${key}: spec.sdk '${spec.sdk}' doesn't match provider sdk '${providerConfig.sdk}'`)
|
|
78
81
|
}
|
package/src/lib/creators.js
CHANGED
|
@@ -65,6 +65,12 @@ const creators = {
|
|
|
65
65
|
logo: 'cohere.svg',
|
|
66
66
|
description: 'Cohere builds retrieval-focused models: embeddings and rerankers aimed at enterprise search rather than chat.'
|
|
67
67
|
},
|
|
68
|
+
typesafe: {
|
|
69
|
+
prefixes: ['jev'],
|
|
70
|
+
label: 'TypeSafe',
|
|
71
|
+
logo: 'typesafe.svg',
|
|
72
|
+
description: 'TypeSafe trains decision models that answer typed questions with calibrated probabilities instead of generating text.'
|
|
73
|
+
},
|
|
68
74
|
nomic: {
|
|
69
75
|
prefixes: ['nomic-embed'],
|
|
70
76
|
label: 'Nomic',
|
package/src/lib/index.js
CHANGED
|
@@ -16,7 +16,7 @@ import {
|
|
|
16
16
|
import { createRateLimiter } from '../../js/session/_rate_limiter.js'
|
|
17
17
|
import { createCooldownTracker } from '../../js/session/_cooldown.js'
|
|
18
18
|
import { setCatalog } from '../../js/session/adapters/_catalog.js'
|
|
19
|
-
import { runAnswer, runAnswerEmbedding, runAnswerImage, runAnswerTranscription } from '../../js/factory/bridge.js'
|
|
19
|
+
import { runAnswer, runAnswerEmbedding, runAnswerEvaluation, runAnswerImage, runAnswerTranscription } from '../../js/factory/bridge.js'
|
|
20
20
|
import { startSpan, endSpanOk, endSpanError } from './tracing.js'
|
|
21
21
|
import { isValidTag } from './schema.js'
|
|
22
22
|
import { silent } from './logger.js'
|
|
@@ -751,6 +751,22 @@ const createModelProxy = (resolvedModelId, modelSpec, handlers, aliasOutputEffor
|
|
|
751
751
|
}
|
|
752
752
|
}
|
|
753
753
|
|
|
754
|
+
if (prop === 'evaluate') {
|
|
755
|
+
return async (state, questions, options = {}) => {
|
|
756
|
+
const { configuration } = await getRuntime()
|
|
757
|
+
return runAnswerEvaluation({
|
|
758
|
+
provider: modelSpec.provider,
|
|
759
|
+
model: modelSpec.model ?? resolvedModelId.split('/').pop(),
|
|
760
|
+
modelKey: resolvedModelId,
|
|
761
|
+
configuration,
|
|
762
|
+
state,
|
|
763
|
+
questions,
|
|
764
|
+
options,
|
|
765
|
+
spec: modelSpec
|
|
766
|
+
}, { limiter: rateLimiter, resolveProviderLimits })
|
|
767
|
+
}
|
|
768
|
+
}
|
|
769
|
+
|
|
754
770
|
if (prop === 'setRateLimit') {
|
|
755
771
|
return async ({ rpm, tpm, inpm } = {}) => {
|
|
756
772
|
const curatedCache = getCuratedCacheSnapshot()
|
package/src/lib/provider-info.js
CHANGED
|
@@ -100,6 +100,13 @@ const PROVIDER_INFO = {
|
|
|
100
100
|
hint: 'Create an API key in the Cohere dashboard under API Keys',
|
|
101
101
|
free: true
|
|
102
102
|
},
|
|
103
|
+
typesafe: {
|
|
104
|
+
label: 'TypeSafe',
|
|
105
|
+
description: 'Calibrated yes/no, choice and score answers about a state. No chat models through mohdel.',
|
|
106
|
+
url: 'https://console.typesafe.ai/keys',
|
|
107
|
+
hint: 'Create an API key in the TypeSafe console under Keys',
|
|
108
|
+
free: false
|
|
109
|
+
},
|
|
103
110
|
qwen: {
|
|
104
111
|
label: 'Qwen Cloud',
|
|
105
112
|
description: 'Qwen — reasoning, coding, long context. Free quota for new users.',
|
package/src/lib/providers.js
CHANGED
|
@@ -1,3 +1,5 @@
|
|
|
1
|
+
import { providerOf } from '#core/model-id.js'
|
|
2
|
+
|
|
1
3
|
const LOCAL_API_KEY_ENV = 'MOHDEL_LOCAL_API_SK'
|
|
2
4
|
|
|
3
5
|
// `contextSemantics` and `outputCapStrategy` are published facts about a
|
|
@@ -6,8 +8,13 @@ const LOCAL_API_KEY_ENV = 'MOHDEL_LOCAL_API_SK'
|
|
|
6
8
|
// own provider requests does not have to rediscover the behaviour one 400 at a
|
|
7
9
|
// time. Entries may override `outputCapStrategy` per model. See
|
|
8
10
|
// ARCHITECTURE.md > "The output budget is capped to the model's ceiling".
|
|
11
|
+
// `billing` is how a provider's calls are paid for. `metered`: API money,
|
|
12
|
+
// reported as `cost`. `plan`: a share of a subscription allowance, seen at
|
|
13
|
+
// `usage`. `capacity`: hardware run or rented at a flat rate. `cost` is 0 for
|
|
14
|
+
// the last two.
|
|
9
15
|
const providers = {
|
|
10
16
|
anthropic: {
|
|
17
|
+
billing: { kind: 'metered' },
|
|
11
18
|
sdk: 'anthropic',
|
|
12
19
|
apiKeyEnv: 'ANTHROPIC_API_SK',
|
|
13
20
|
createConfiguration: apiKey => ({ apiKey }),
|
|
@@ -20,6 +27,7 @@ const providers = {
|
|
|
20
27
|
outputCapStrategy: 'error'
|
|
21
28
|
},
|
|
22
29
|
cerebras: {
|
|
30
|
+
billing: { kind: 'metered' },
|
|
23
31
|
sdk: 'cerebras',
|
|
24
32
|
apiKeyEnv: 'CEREBRAS_API_SK',
|
|
25
33
|
createConfiguration: apiKey => ({ apiKey }),
|
|
@@ -32,6 +40,7 @@ const providers = {
|
|
|
32
40
|
outputCapStrategy: 'accept'
|
|
33
41
|
},
|
|
34
42
|
chatgpt: {
|
|
43
|
+
billing: { kind: 'plan', label: 'ChatGPT plan', usage: 'https://chatgpt.com/settings/usage' },
|
|
35
44
|
sdk: 'openai',
|
|
36
45
|
catalogClient: 'chatgpt',
|
|
37
46
|
refreshConfiguration: true,
|
|
@@ -47,6 +56,7 @@ const providers = {
|
|
|
47
56
|
outputCapStrategy: 'accept'
|
|
48
57
|
},
|
|
49
58
|
deepseek: {
|
|
59
|
+
billing: { kind: 'metered' },
|
|
50
60
|
sdk: 'openai',
|
|
51
61
|
api: 'chatCompletions',
|
|
52
62
|
apiKeyEnv: 'DEEPSEEK_API_SK',
|
|
@@ -61,6 +71,7 @@ const providers = {
|
|
|
61
71
|
outputCapStrategy: 'accept'
|
|
62
72
|
},
|
|
63
73
|
fireworks: {
|
|
74
|
+
billing: { kind: 'metered' },
|
|
64
75
|
sdk: 'fireworks',
|
|
65
76
|
apiKeyEnv: 'FIREWORKS_API_SK',
|
|
66
77
|
baseURL: 'https://api.fireworks.ai/inference/v1',
|
|
@@ -74,6 +85,7 @@ const providers = {
|
|
|
74
85
|
outputCapStrategy: 'accept'
|
|
75
86
|
},
|
|
76
87
|
gemini: {
|
|
88
|
+
billing: { kind: 'metered' },
|
|
77
89
|
sdk: 'gemini',
|
|
78
90
|
apiKeyEnv: 'GEMINI_API_SK',
|
|
79
91
|
createConfiguration: apiKey => ({ apiKey }),
|
|
@@ -86,6 +98,7 @@ const providers = {
|
|
|
86
98
|
outputCapStrategy: 'accept'
|
|
87
99
|
},
|
|
88
100
|
groq: {
|
|
101
|
+
billing: { kind: 'metered' },
|
|
89
102
|
sdk: 'groq',
|
|
90
103
|
apiKeyEnv: 'GROQ_API_SK',
|
|
91
104
|
createConfiguration: apiKey => ({ apiKey }),
|
|
@@ -96,6 +109,7 @@ const providers = {
|
|
|
96
109
|
}
|
|
97
110
|
},
|
|
98
111
|
local: {
|
|
112
|
+
billing: { kind: 'capacity', label: 'local server' },
|
|
99
113
|
sdk: 'openai',
|
|
100
114
|
api: 'chatCompletions',
|
|
101
115
|
catalog: false,
|
|
@@ -105,6 +119,7 @@ const providers = {
|
|
|
105
119
|
outputCapStrategy: 'accept'
|
|
106
120
|
},
|
|
107
121
|
meta: {
|
|
122
|
+
billing: { kind: 'metered' },
|
|
108
123
|
sdk: 'openai',
|
|
109
124
|
apiKeyEnv: 'META_API_SK',
|
|
110
125
|
baseURL: 'https://api.meta.ai/v1',
|
|
@@ -117,6 +132,7 @@ const providers = {
|
|
|
117
132
|
outputCapStrategy: 'accept'
|
|
118
133
|
},
|
|
119
134
|
cohere: {
|
|
135
|
+
billing: { kind: 'metered' },
|
|
120
136
|
sdk: 'cohere',
|
|
121
137
|
api: 'embeddings',
|
|
122
138
|
apiKeyEnv: 'COHERE_API_SK',
|
|
@@ -128,7 +144,22 @@ const providers = {
|
|
|
128
144
|
rateLimits: 'https://docs.cohere.com/docs/rate-limits'
|
|
129
145
|
}
|
|
130
146
|
},
|
|
147
|
+
typesafe: {
|
|
148
|
+
billing: { kind: 'metered' },
|
|
149
|
+
sdk: 'typesafe',
|
|
150
|
+
api: 'evaluation',
|
|
151
|
+
catalog: false,
|
|
152
|
+
apiKeyEnv: 'TYPESAFE_API_SK',
|
|
153
|
+
baseURL: 'https://api.typesafe.ai/v1',
|
|
154
|
+
createConfiguration: apiKey => ({ apiKey }),
|
|
155
|
+
references: {
|
|
156
|
+
pricing: 'https://docs.typesafe.ai/models',
|
|
157
|
+
models: 'https://docs.typesafe.ai/models',
|
|
158
|
+
rateLimits: 'https://docs.typesafe.ai/models'
|
|
159
|
+
}
|
|
160
|
+
},
|
|
131
161
|
mistral: {
|
|
162
|
+
billing: { kind: 'metered' },
|
|
132
163
|
sdk: 'openai',
|
|
133
164
|
api: 'chatCompletions',
|
|
134
165
|
apiKeyEnv: 'MISTRAL_API_SK',
|
|
@@ -141,6 +172,7 @@ const providers = {
|
|
|
141
172
|
}
|
|
142
173
|
},
|
|
143
174
|
novita: {
|
|
175
|
+
billing: { kind: 'metered' },
|
|
144
176
|
sdk: 'openai',
|
|
145
177
|
api: 'chatCompletions',
|
|
146
178
|
imageHandler: 'novita',
|
|
@@ -157,6 +189,7 @@ const providers = {
|
|
|
157
189
|
outputCapStrategy: 'error'
|
|
158
190
|
},
|
|
159
191
|
openai: {
|
|
192
|
+
billing: { kind: 'metered' },
|
|
160
193
|
sdk: 'openai',
|
|
161
194
|
apiKeyEnv: 'OPENAI_API_SK',
|
|
162
195
|
createConfiguration: apiKey => ({ apiKey }),
|
|
@@ -169,6 +202,7 @@ const providers = {
|
|
|
169
202
|
outputCapStrategy: 'accept'
|
|
170
203
|
},
|
|
171
204
|
openrouter: {
|
|
205
|
+
billing: { kind: 'metered' },
|
|
172
206
|
sdk: 'openrouter',
|
|
173
207
|
apiKeyEnv: 'OPENROUTER_API_SK',
|
|
174
208
|
baseURL: 'https://openrouter.ai/api/v1',
|
|
@@ -184,6 +218,7 @@ const providers = {
|
|
|
184
218
|
}
|
|
185
219
|
},
|
|
186
220
|
qwen: {
|
|
221
|
+
billing: { kind: 'metered' },
|
|
187
222
|
sdk: 'openai',
|
|
188
223
|
api: 'chatCompletions',
|
|
189
224
|
apiKeyEnv: 'QWEN_API_SK',
|
|
@@ -198,6 +233,7 @@ const providers = {
|
|
|
198
233
|
outputCapStrategy: 'accept'
|
|
199
234
|
},
|
|
200
235
|
xai: {
|
|
236
|
+
billing: { kind: 'metered' },
|
|
201
237
|
sdk: 'openai',
|
|
202
238
|
apiKeyEnv: 'XAI_API_SK',
|
|
203
239
|
baseURL: 'https://api.x.ai/v1',
|
|
@@ -211,6 +247,7 @@ const providers = {
|
|
|
211
247
|
outputCapStrategy: 'accept'
|
|
212
248
|
},
|
|
213
249
|
xiaomi: {
|
|
250
|
+
billing: { kind: 'metered' },
|
|
214
251
|
sdk: 'openai',
|
|
215
252
|
api: 'chatCompletions',
|
|
216
253
|
apiKeyEnv: 'XIAOMI_API_SK',
|
|
@@ -223,4 +260,16 @@ const providers = {
|
|
|
223
260
|
|
|
224
261
|
Object.freeze(providers)
|
|
225
262
|
|
|
263
|
+
/**
|
|
264
|
+
* How calls to `modelId`'s provider are paid for.
|
|
265
|
+
* @param {string} modelId
|
|
266
|
+
* @returns {{kind: 'metered' | 'plan' | 'capacity', label?: string, usage?: string}}
|
|
267
|
+
*/
|
|
268
|
+
export function billingOf (modelId) {
|
|
269
|
+
const provider = providerOf(modelId)
|
|
270
|
+
const def = providers[provider]
|
|
271
|
+
if (!def) throw new Error(`Unknown provider '${provider}' in model id '${modelId}'`)
|
|
272
|
+
return { ...def.billing }
|
|
273
|
+
}
|
|
274
|
+
|
|
226
275
|
export default providers
|
package/src/lib/schema.js
CHANGED
|
@@ -65,6 +65,7 @@ const fieldDefs = {
|
|
|
65
65
|
maxInputTokens: { type: 'number' },
|
|
66
66
|
inputTypes: { type: 'object' },
|
|
67
67
|
defaultInputType: { type: 'string' },
|
|
68
|
+
evaluationTypes: { type: 'array', itemType: 'string' },
|
|
68
69
|
deprecated: { type: 'string' },
|
|
69
70
|
suspended: { type: 'string' },
|
|
70
71
|
rpmLimit: { type: 'number' },
|