dsh-autotier 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +93 -0
- package/CHANGELOG.md +85 -0
- package/LICENSE +201 -0
- package/README.es.md +247 -0
- package/README.hi.md +241 -0
- package/README.md +245 -0
- package/README.pt.md +246 -0
- package/README.zh.md +221 -0
- package/SECURITY.md +55 -0
- package/THIRD_PARTY_NOTICES.md +63 -0
- package/cordis.patch.yml +125 -0
- package/docs/preset-row.md +61 -0
- package/docs/supporting-lanes.md +45 -0
- package/lib/index.js +2848 -0
- package/lib/types/command.d.ts +17 -0
- package/lib/types/command.d.ts.map +1 -0
- package/lib/types/config.d.ts +94 -0
- package/lib/types/config.d.ts.map +1 -0
- package/lib/types/guard-rules.d.ts +97 -0
- package/lib/types/guard-rules.d.ts.map +1 -0
- package/lib/types/guard.d.ts +70 -0
- package/lib/types/guard.d.ts.map +1 -0
- package/lib/types/index.d.ts +60 -0
- package/lib/types/index.d.ts.map +1 -0
- package/lib/types/intent.d.ts +179 -0
- package/lib/types/intent.d.ts.map +1 -0
- package/lib/types/judge.d.ts +50 -0
- package/lib/types/judge.d.ts.map +1 -0
- package/lib/types/policy.d.ts +109 -0
- package/lib/types/policy.d.ts.map +1 -0
- package/lib/types/routing.d.ts +135 -0
- package/lib/types/routing.d.ts.map +1 -0
- package/lib/types/schema.d.ts +134 -0
- package/lib/types/schema.d.ts.map +1 -0
- package/lib/types/service.d.ts +67 -0
- package/lib/types/service.d.ts.map +1 -0
- package/lib/types/state.d.ts +46 -0
- package/lib/types/state.d.ts.map +1 -0
- package/lib/types/tiers.d.ts +103 -0
- package/lib/types/tiers.d.ts.map +1 -0
- package/lib/types/tools.d.ts +26 -0
- package/lib/types/tools.d.ts.map +1 -0
- package/lib/types/types.d.ts +96 -0
- package/lib/types/types.d.ts.map +1 -0
- package/package.json +179 -0
- package/src/command.ts +73 -0
- package/src/config.ts +358 -0
- package/src/guard-rules.ts +303 -0
- package/src/guard.ts +285 -0
- package/src/index.ts +149 -0
- package/src/intent.ts +484 -0
- package/src/judge.ts +150 -0
- package/src/policy.ts +246 -0
- package/src/routing.ts +575 -0
- package/src/schema.ts +295 -0
- package/src/service.ts +131 -0
- package/src/state.ts +134 -0
- package/src/tiers.ts +212 -0
- package/src/tools.ts +128 -0
- package/src/types.ts +120 -0
package/src/tiers.ts
ADDED
|
@@ -0,0 +1,212 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Pure tier arithmetic: the adapter-owned effort ladder, tier-route
|
|
3
|
+
* application, the effort-first escalation ladder, and the fallback-chain
|
|
4
|
+
* classifier/advance helpers.
|
|
5
|
+
*
|
|
6
|
+
* The fallback vocabulary is ported from `dsh-tier-router`'s `lib/pure.js`
|
|
7
|
+
* (MIT; see `THIRD_PARTY_NOTICES.md`), with two corrections: the effort ladder
|
|
8
|
+
* is DeepSeek's `off | low | high | max` (upstream's `medium` does not exist and
|
|
9
|
+
* would fail every request with `UNSUPPORTED_REASONING_EFFORT`), and the
|
|
10
|
+
* classification is split into permanent/transient so the request-error handler
|
|
11
|
+
* can honour the division of labour with `dsh-llm-retry` (permanent codes switch
|
|
12
|
+
* the chain immediately; transient codes wait for retry exhaustion).
|
|
13
|
+
*
|
|
14
|
+
* @module dsh-autotier/tiers
|
|
15
|
+
*/
|
|
16
|
+
|
|
17
|
+
import type { LlmCallConfig } from '@deepseek-ai/dsh-llm'
|
|
18
|
+
import type { EffortId, TierId, TierRoute } from './types.ts'
|
|
19
|
+
|
|
20
|
+
/** The adapter-owned effort ladder, cheapest to strongest. */
|
|
21
|
+
export const EFFORT_LADDER = ['off', 'low', 'high', 'max'] as const satisfies readonly EffortId[]
|
|
22
|
+
|
|
23
|
+
/** Position of one effort on the ladder (0..3); unknown ids rank lowest. */
|
|
24
|
+
export function effortRank(effort: string): number {
|
|
25
|
+
const index = (EFFORT_LADDER as readonly string[]).indexOf(effort)
|
|
26
|
+
return index === -1 ? 0 : index
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
/**
|
|
30
|
+
* The next effort step above `current`, never above `ceiling`.
|
|
31
|
+
* @param current - the effort currently in force.
|
|
32
|
+
* @param ceiling - the strongest effort to consider (default `max`).
|
|
33
|
+
* @returns the next effort id, or null when already at or above the ceiling.
|
|
34
|
+
*/
|
|
35
|
+
export function nextEffortStep(current: EffortId, ceiling: EffortId = 'max'): EffortId | null {
|
|
36
|
+
const start = effortRank(current)
|
|
37
|
+
const stop = effortRank(ceiling)
|
|
38
|
+
if (start >= stop) return null
|
|
39
|
+
return EFFORT_LADDER[start + 1] ?? null
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
/** Whether two routes land on the same provider/model/effort triple. */
|
|
43
|
+
export function routeEquals(a: TierRoute, b: TierRoute): boolean {
|
|
44
|
+
return a.provider === b.provider && a.model === b.model && a.effort === b.effort
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
/** One adapter-owned effort id as the LLM call config expects it (branded at the seam). */
|
|
48
|
+
function brandedEffort(effort: string): NonNullable<LlmCallConfig['reasoningEffort']> {
|
|
49
|
+
// The adapter's `ReasoningEffortId` is a branded string; the brand is opaque
|
|
50
|
+
// by design, and our vocabulary is validated against it before it gets here.
|
|
51
|
+
return effort as unknown as NonNullable<LlmCallConfig['reasoningEffort']>
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
/**
|
|
55
|
+
* Apply one tier route to a request configuration. The sampling scalars the
|
|
56
|
+
* session already chose (`temperature`, `maxTokens`, `stop`) are preserved
|
|
57
|
+
* exactly; the provider/model/effort triple is replaced. Returns the input
|
|
58
|
+
* object unchanged when the route already matches, so the caller can skip a
|
|
59
|
+
* logged header change.
|
|
60
|
+
*
|
|
61
|
+
* @param base - the configuration the loop proposed.
|
|
62
|
+
* @param target - the tier landing to apply.
|
|
63
|
+
* @returns the replacement configuration (or `base` when identical).
|
|
64
|
+
*/
|
|
65
|
+
export function resolveRoute(base: LlmCallConfig, target: TierRoute): LlmCallConfig {
|
|
66
|
+
const sameLanding = base.provider === target.provider && base.model === target.model
|
|
67
|
+
if (sameLanding && (target.effort === undefined || base.reasoningEffort === target.effort)) return base
|
|
68
|
+
const next: LlmCallConfig = { provider: target.provider, model: target.model }
|
|
69
|
+
// An absent target effort means "follow the session": carry the effort the
|
|
70
|
+
// request already carries instead of dropping it (a dropped effort silently
|
|
71
|
+
// falls back to the adapter default, which is the strongest one).
|
|
72
|
+
const effort = target.effort ?? base.reasoningEffort
|
|
73
|
+
if (effort !== undefined) next.reasoningEffort = brandedEffort(effort)
|
|
74
|
+
if (base.temperature !== undefined) next.temperature = base.temperature
|
|
75
|
+
if (base.maxTokens !== undefined) next.maxTokens = base.maxTokens
|
|
76
|
+
if (base.stop !== undefined) next.stop = base.stop
|
|
77
|
+
return next
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
/** One rung of the effort-first escalation ladder. */
|
|
81
|
+
export interface EscalationRung {
|
|
82
|
+
/** The landing this rung applies. */
|
|
83
|
+
readonly route: TierRoute
|
|
84
|
+
/** The tier the rung belongs to (an effort rung on the cheap model is still cheap). */
|
|
85
|
+
readonly tier: TierId
|
|
86
|
+
/** Why this rung exists, for logs and `/tier status`. */
|
|
87
|
+
readonly note: string
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
/**
|
|
91
|
+
* Build the effort-first escalation ladder: raise the current model's effort
|
|
92
|
+
* one step at a time (the KV prefix survives, and the official model-selection
|
|
93
|
+
* notice is not emitted for an effort-only change) before paying for a model
|
|
94
|
+
* switch. When both tiers share one model the ladder collapses to a single
|
|
95
|
+
* effort-only rung.
|
|
96
|
+
*
|
|
97
|
+
* @param base - the configuration the loop proposed for the failing step.
|
|
98
|
+
* @param cheap - the cheap tier landing.
|
|
99
|
+
* @param strong - the strong tier landing.
|
|
100
|
+
* @returns the ordered rungs; empty when escalation cannot change anything.
|
|
101
|
+
*/
|
|
102
|
+
export function escalationLadder(base: LlmCallConfig, cheap: TierRoute, strong: TierRoute): EscalationRung[] {
|
|
103
|
+
const rungs: EscalationRung[] = []
|
|
104
|
+
const baseEffort = (base.reasoningEffort ?? cheap.effort ?? 'low') as EffortId
|
|
105
|
+
const onCheapModel = base.provider === cheap.provider && base.model === cheap.model
|
|
106
|
+
const sharedModel = strong.provider === cheap.provider && strong.model === cheap.model
|
|
107
|
+
if (sharedModel) {
|
|
108
|
+
if (strong.effort !== undefined && strong.effort !== baseEffort) {
|
|
109
|
+
rungs.push({
|
|
110
|
+
route: { provider: strong.provider, model: strong.model, effort: strong.effort },
|
|
111
|
+
tier: 'strong',
|
|
112
|
+
note: `effort-only escalation on the shared model (${baseEffort} -> ${strong.effort})`,
|
|
113
|
+
})
|
|
114
|
+
}
|
|
115
|
+
return rungs
|
|
116
|
+
}
|
|
117
|
+
if (onCheapModel) {
|
|
118
|
+
let step = nextEffortStep(baseEffort)
|
|
119
|
+
while (step !== null) {
|
|
120
|
+
rungs.push({
|
|
121
|
+
route: { provider: cheap.provider, model: cheap.model, effort: step },
|
|
122
|
+
tier: 'cheap',
|
|
123
|
+
note: `cheap-tier effort rung ${step}`,
|
|
124
|
+
})
|
|
125
|
+
step = nextEffortStep(step)
|
|
126
|
+
}
|
|
127
|
+
}
|
|
128
|
+
rungs.push({
|
|
129
|
+
route: strong.effort === undefined
|
|
130
|
+
? { provider: strong.provider, model: strong.model }
|
|
131
|
+
: { provider: strong.provider, model: strong.model, effort: strong.effort },
|
|
132
|
+
tier: 'strong',
|
|
133
|
+
note: 'switch to the strong model',
|
|
134
|
+
})
|
|
135
|
+
return rungs
|
|
136
|
+
}
|
|
137
|
+
|
|
138
|
+
/** Failure codes that mean the route itself is unusable: switch the chain now. */
|
|
139
|
+
export const FALLBACK_PERMANENT_CODES = [
|
|
140
|
+
'UNKNOWN_MODEL',
|
|
141
|
+
'MISSING_CREDENTIAL',
|
|
142
|
+
'INVALID_CREDENTIAL',
|
|
143
|
+
'QUOTA',
|
|
144
|
+
] as const
|
|
145
|
+
|
|
146
|
+
/** Failure codes owned by `dsh-llm-retry` first: switch the chain only after retries are exhausted. */
|
|
147
|
+
export const FALLBACK_TRANSIENT_CODES = ['RATE_LIMIT', 'SERVER', 'TIMEOUT', 'TRANSPORT'] as const
|
|
148
|
+
|
|
149
|
+
/** Failure codes that must never switch the model. */
|
|
150
|
+
export const FALLBACK_IGNORE_CODES = [
|
|
151
|
+
'CONTEXT_WINDOW_EXCEEDED',
|
|
152
|
+
'UNSUPPORTED_REASONING_EFFORT',
|
|
153
|
+
'ABORTED',
|
|
154
|
+
'EMPTY_RESPONSE',
|
|
155
|
+
] as const
|
|
156
|
+
|
|
157
|
+
/** How one failure relates to the fallback chain. */
|
|
158
|
+
export type FallbackClass = 'permanent' | 'transient' | 'ignore' | 'unknown'
|
|
159
|
+
|
|
160
|
+
/**
|
|
161
|
+
* Classify one failed model request for the fallback machinery.
|
|
162
|
+
* @param failure - the normalized failure facts (`code` and optional `status`).
|
|
163
|
+
* @returns the chain verdict.
|
|
164
|
+
*/
|
|
165
|
+
export function classifyFallback(failure: { code?: unknown; status?: unknown } | undefined): FallbackClass {
|
|
166
|
+
if (failure === undefined || failure === null || typeof failure !== 'object') return 'unknown'
|
|
167
|
+
const code = typeof failure.code === 'string' ? failure.code : undefined
|
|
168
|
+
if (code !== undefined) {
|
|
169
|
+
if ((FALLBACK_PERMANENT_CODES as readonly string[]).includes(code)) return 'permanent'
|
|
170
|
+
if ((FALLBACK_TRANSIENT_CODES as readonly string[]).includes(code)) return 'transient'
|
|
171
|
+
if ((FALLBACK_IGNORE_CODES as readonly string[]).includes(code)) return 'ignore'
|
|
172
|
+
}
|
|
173
|
+
if (typeof failure.status === 'number' && failure.status >= 500) return 'transient'
|
|
174
|
+
return 'unknown'
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
/** One agent's position in one tier's fallback chain. */
|
|
178
|
+
export interface FallbackRecord {
|
|
179
|
+
/** The tier whose chain this record belongs to. */
|
|
180
|
+
tier: TierId
|
|
181
|
+
/** Index in that tier's chain; -1 means the tier's own landing. */
|
|
182
|
+
index: number
|
|
183
|
+
/** Epoch millis until which the record stays in force. */
|
|
184
|
+
until: number
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
/** Whether a fallback record is currently pinning the agent to a chain entry. */
|
|
188
|
+
export function fallbackActive(record: FallbackRecord | undefined, now: number): boolean {
|
|
189
|
+
return record !== undefined && record.index >= 0 && record.until > now
|
|
190
|
+
}
|
|
191
|
+
|
|
192
|
+
/**
|
|
193
|
+
* Advance a fallback record one step down one tier's chain.
|
|
194
|
+
* @param record - the current record (absent or from another tier = start of this chain).
|
|
195
|
+
* @param tier - the tier whose chain is being walked.
|
|
196
|
+
* @param chainLength - the number of configured fallback entries.
|
|
197
|
+
* @param now - current epoch millis.
|
|
198
|
+
* @param ttlMs - how long the new entry stays in force.
|
|
199
|
+
* @returns the next record, or null when the chain is exhausted.
|
|
200
|
+
*/
|
|
201
|
+
export function advanceFallback(
|
|
202
|
+
record: FallbackRecord | undefined,
|
|
203
|
+
tier: TierId,
|
|
204
|
+
chainLength: number,
|
|
205
|
+
now: number,
|
|
206
|
+
ttlMs: number,
|
|
207
|
+
): FallbackRecord | null {
|
|
208
|
+
const index = record !== undefined && record.tier === tier ? record.index : -1
|
|
209
|
+
const next = index + 1
|
|
210
|
+
if (next >= chainLength) return null
|
|
211
|
+
return { tier, index: next, until: now + (Number.isFinite(ttlMs) && ttlMs > 0 ? ttlMs : 300_000) }
|
|
212
|
+
}
|
package/src/tools.ts
ADDED
|
@@ -0,0 +1,128 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The two read-only tools: `tier_status` reports the live routing state, and
|
|
3
|
+
* `tier_route` dry-runs the classifier on one intent string without sending a
|
|
4
|
+
* model request.
|
|
5
|
+
* @module dsh-autotier/tools
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
import type { Context } from '@deepseek-ai/cordis'
|
|
9
|
+
import type { Agent } from '@deepseek-ai/dsh-agent'
|
|
10
|
+
import { defineTool, type ToolDefinition } from '@deepseek-ai/dsh-tools'
|
|
11
|
+
import { classifyIntent } from './intent.ts'
|
|
12
|
+
import { escalationActive } from './policy.ts'
|
|
13
|
+
import type { AutotierService } from './service.ts'
|
|
14
|
+
import type { AgentStateStore } from './state.ts'
|
|
15
|
+
|
|
16
|
+
/** Services the tools read. */
|
|
17
|
+
export interface ToolServices {
|
|
18
|
+
readonly service: AutotierService
|
|
19
|
+
readonly states: AgentStateStore
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
/** Render one canonical value as a single text block. */
|
|
23
|
+
function textBlock(value: unknown): { type: 'text'; text: string }[] {
|
|
24
|
+
return [{ type: 'text', text: typeof value === 'string' ? value : JSON.stringify(value, null, 2) }]
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
/** Build the `tier_status` tool. */
|
|
28
|
+
export function tierStatusTool({ service, states }: ToolServices): ToolDefinition {
|
|
29
|
+
return defineTool({
|
|
30
|
+
name: 'tier_status',
|
|
31
|
+
description:
|
|
32
|
+
'Report the live dsh-autotier routing state: the configured tier landings, the effective mode for this session, the last intent classification, and whether a failure escalation or fallback is in force. Read-only.',
|
|
33
|
+
parameters: {},
|
|
34
|
+
output: {
|
|
35
|
+
schema: {
|
|
36
|
+
type: 'object',
|
|
37
|
+
additionalProperties: false,
|
|
38
|
+
properties: {
|
|
39
|
+
mode: { type: 'string' },
|
|
40
|
+
strong: { type: 'string' },
|
|
41
|
+
cheap: { type: 'string' },
|
|
42
|
+
vision: { type: 'string' },
|
|
43
|
+
guardEnabled: { type: 'boolean' },
|
|
44
|
+
appliedTier: { type: 'string' },
|
|
45
|
+
lastScenario: { type: 'string' },
|
|
46
|
+
lastConfidence: { type: 'number' },
|
|
47
|
+
escalated: { type: 'boolean' },
|
|
48
|
+
planActive: { type: 'boolean' },
|
|
49
|
+
guardDenials: { type: 'integer' },
|
|
50
|
+
lastDenialRule: { type: 'string' },
|
|
51
|
+
},
|
|
52
|
+
},
|
|
53
|
+
render: (_args, value) => textBlock(value),
|
|
54
|
+
},
|
|
55
|
+
execute: async (_args, exec) => {
|
|
56
|
+
const agent = exec.agent as Agent | undefined
|
|
57
|
+
const state = agent === undefined ? undefined : states.for(agent)
|
|
58
|
+
const status = service.status()
|
|
59
|
+
return {
|
|
60
|
+
mode: state?.override ?? status.mode,
|
|
61
|
+
strong: `${status.tiers.strong.provider}/${status.tiers.strong.model}${status.tiers.strong.effort === undefined ? '' : `@${status.tiers.strong.effort}`}`,
|
|
62
|
+
cheap: `${status.tiers.cheap.provider}/${status.tiers.cheap.model}${status.tiers.cheap.effort === undefined ? '' : `@${status.tiers.cheap.effort}`}`,
|
|
63
|
+
vision: `${status.tiers.vision.provider}/${status.tiers.vision.model}`,
|
|
64
|
+
guardEnabled: status.guard.enabled,
|
|
65
|
+
appliedTier: state?.appliedTier ?? '',
|
|
66
|
+
lastScenario: state?.decision?.scenario ?? '',
|
|
67
|
+
lastConfidence: state?.decision?.confidence ?? 0,
|
|
68
|
+
escalated: state === undefined ? false : escalationActive(state, Date.now()),
|
|
69
|
+
planActive: state?.planActive ?? false,
|
|
70
|
+
guardDenials: state?.denials ?? 0,
|
|
71
|
+
lastDenialRule: state?.lastDenial ?? '',
|
|
72
|
+
}
|
|
73
|
+
},
|
|
74
|
+
})
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
/** Build the `tier_route` tool. */
|
|
78
|
+
export function tierRouteTool({ service }: ToolServices): ToolDefinition {
|
|
79
|
+
return defineTool({
|
|
80
|
+
name: 'tier_route',
|
|
81
|
+
description:
|
|
82
|
+
'Dry-run the intent classifier on one instruction and report which tier it would use, without sending a model request. Read-only.',
|
|
83
|
+
parameters: {
|
|
84
|
+
text: { type: 'string', description: 'The instruction to classify.', required: true as const },
|
|
85
|
+
},
|
|
86
|
+
output: {
|
|
87
|
+
schema: {
|
|
88
|
+
type: 'object',
|
|
89
|
+
additionalProperties: false,
|
|
90
|
+
properties: {
|
|
91
|
+
tier: { type: 'string' },
|
|
92
|
+
scenario: { type: 'string' },
|
|
93
|
+
confidence: { type: 'number' },
|
|
94
|
+
fingerprint: { type: 'string' },
|
|
95
|
+
reason: { type: 'string' },
|
|
96
|
+
},
|
|
97
|
+
},
|
|
98
|
+
render: (_args, value) => textBlock(value),
|
|
99
|
+
},
|
|
100
|
+
execute: async (args) => {
|
|
101
|
+
const config = service.config()
|
|
102
|
+
const result = classifyIntent({
|
|
103
|
+
text: args.text,
|
|
104
|
+
toolNames: [],
|
|
105
|
+
hasImage: false,
|
|
106
|
+
messageCount: 0,
|
|
107
|
+
cwd: '',
|
|
108
|
+
}, { rules: service.rules(), scenarios: config.intent.scenarios })
|
|
109
|
+
return {
|
|
110
|
+
tier: result.tier,
|
|
111
|
+
scenario: result.scenario,
|
|
112
|
+
confidence: result.confidence,
|
|
113
|
+
fingerprint: result.fingerprint,
|
|
114
|
+
reason: result.reasons.join('; '),
|
|
115
|
+
}
|
|
116
|
+
},
|
|
117
|
+
})
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
/**
|
|
121
|
+
* Register both tools on the plugin fiber.
|
|
122
|
+
* @param ctx - the plugin context (must have `tools`).
|
|
123
|
+
* @param services - the service and the state store.
|
|
124
|
+
*/
|
|
125
|
+
export function registerTierTools(ctx: Context, services: ToolServices): void {
|
|
126
|
+
ctx.tools.register(tierStatusTool(services))
|
|
127
|
+
ctx.tools.register(tierRouteTool(services))
|
|
128
|
+
}
|
package/src/types.ts
ADDED
|
@@ -0,0 +1,120 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Shared vocabulary for dsh-autotier: the tier ids, the adapter-owned reasoning
|
|
3
|
+
* effort ids, the routing modes, and the public route/status shapes every
|
|
4
|
+
* module and the `ctx.autotier` service speak.
|
|
5
|
+
* @module dsh-autotier/types
|
|
6
|
+
*/
|
|
7
|
+
|
|
8
|
+
/**
|
|
9
|
+
* Service Definition (contract layer): the two cost tiers this plugin routes
|
|
10
|
+
* between. `strong` plans complex intent and reviews high-risk work; `cheap`
|
|
11
|
+
* implements it. The ids are stable wire/settings vocabulary.
|
|
12
|
+
*/
|
|
13
|
+
export const TIER_IDS = ['strong', 'cheap'] as const
|
|
14
|
+
|
|
15
|
+
/** One of the two cost tiers. */
|
|
16
|
+
export type TierId = (typeof TIER_IDS)[number]
|
|
17
|
+
|
|
18
|
+
/**
|
|
19
|
+
* Reasoning-effort ids owned by the provider adapter. The DeepSeek adapter
|
|
20
|
+
* accepts exactly these four (`packages/llm/llm-deepseek/src/index.ts:133`);
|
|
21
|
+
* `medium` does not exist, and an unsupported value fails every request with
|
|
22
|
+
* `UNSUPPORTED_REASONING_EFFORT` instead of degrading.
|
|
23
|
+
*/
|
|
24
|
+
export const EFFORT_IDS = ['off', 'low', 'high', 'max'] as const
|
|
25
|
+
|
|
26
|
+
/** One adapter-owned reasoning-effort id. */
|
|
27
|
+
export type EffortId = (typeof EFFORT_IDS)[number]
|
|
28
|
+
|
|
29
|
+
/**
|
|
30
|
+
* Routing modes. `auto` is the fully automatic path; the rest are explicit
|
|
31
|
+
* user overrides (`/tier strong|cheap|off`). `delegated` means the session
|
|
32
|
+
* carries an explicit model selection that autotier must not fight.
|
|
33
|
+
*/
|
|
34
|
+
export const ROUTING_MODES = ['auto', 'strong', 'cheap', 'delegated', 'off'] as const
|
|
35
|
+
|
|
36
|
+
/** One routing mode. */
|
|
37
|
+
export type RoutingMode = (typeof ROUTING_MODES)[number]
|
|
38
|
+
|
|
39
|
+
/** Intent classes the rule layer and the judge distinguish. */
|
|
40
|
+
export const SCENARIOS = [
|
|
41
|
+
'coding',
|
|
42
|
+
'review',
|
|
43
|
+
'planning',
|
|
44
|
+
'retrieval',
|
|
45
|
+
'batch',
|
|
46
|
+
'daily',
|
|
47
|
+
'longText',
|
|
48
|
+
'multimodal',
|
|
49
|
+
] as const
|
|
50
|
+
|
|
51
|
+
/** One intent class. */
|
|
52
|
+
export type Scenario = (typeof SCENARIOS)[number]
|
|
53
|
+
|
|
54
|
+
/** Cost/quality arbitration direction when the signals are ambiguous. */
|
|
55
|
+
export const COST_MODES = ['cost-first', 'quality-first', 'balanced'] as const
|
|
56
|
+
|
|
57
|
+
/** One cost/quality arbitration direction. */
|
|
58
|
+
export type CostMode = (typeof COST_MODES)[number]
|
|
59
|
+
|
|
60
|
+
/**
|
|
61
|
+
* One concrete tier landing: the provider/model pair plus the reasoning effort
|
|
62
|
+
* the request should carry. `effort` is omitted only when the tier inherits
|
|
63
|
+
* the session's own effort (followSession).
|
|
64
|
+
*/
|
|
65
|
+
export interface TierRoute {
|
|
66
|
+
readonly provider: string
|
|
67
|
+
readonly model: string
|
|
68
|
+
readonly effort?: EffortId
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
/**
|
|
72
|
+
* Why a tier was chosen. `source` is the arbitration layer that won, so an
|
|
73
|
+
* operator can tell a rule hit from a judge call from an escalation.
|
|
74
|
+
*/
|
|
75
|
+
export type RouteSource =
|
|
76
|
+
| 'manual'
|
|
77
|
+
| 'rule'
|
|
78
|
+
| 'judge'
|
|
79
|
+
| 'posterior'
|
|
80
|
+
| 'plan-mode'
|
|
81
|
+
| 'guard'
|
|
82
|
+
| 'escalation'
|
|
83
|
+
| 'fallback'
|
|
84
|
+
| 'default'
|
|
85
|
+
|
|
86
|
+
/** One routing decision with its provenance. */
|
|
87
|
+
export interface RouteDecision {
|
|
88
|
+
/** The tier the request should use. */
|
|
89
|
+
readonly tier: TierId
|
|
90
|
+
/** The concrete landing. */
|
|
91
|
+
readonly route: TierRoute
|
|
92
|
+
/** The arbitration layer that produced this decision. */
|
|
93
|
+
readonly source: RouteSource
|
|
94
|
+
/** 0..1 confidence of the intent classification (1 for explicit overrides). */
|
|
95
|
+
readonly confidence: number
|
|
96
|
+
/** Short human-readable reason, safe for logs and `/tier status`. */
|
|
97
|
+
readonly reason: string
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
/**
|
|
101
|
+
* Service Definition (contract layer): the read-only status snapshot served by
|
|
102
|
+
* `ctx.autotier.status()` and rendered by `/tier status` and the `tier_status`
|
|
103
|
+
* tool. Third-party plugins may depend on this shape.
|
|
104
|
+
*/
|
|
105
|
+
export interface AutotierStatus {
|
|
106
|
+
/** Current routing mode. */
|
|
107
|
+
readonly mode: RoutingMode
|
|
108
|
+
/** Configured landing per tier (effort omitted when inherited). */
|
|
109
|
+
readonly tiers: { readonly strong: TierRoute; readonly cheap: TierRoute; readonly vision: TierRoute }
|
|
110
|
+
/** Guard switches. */
|
|
111
|
+
readonly guard: { readonly enabled: boolean; readonly tiers: readonly TierId[] }
|
|
112
|
+
/** Escalation switches. */
|
|
113
|
+
readonly escalation: {
|
|
114
|
+
readonly threshold: number
|
|
115
|
+
readonly windowMs: number
|
|
116
|
+
readonly ttlMs: number
|
|
117
|
+
readonly fallbackTtlMs: number
|
|
118
|
+
readonly signature: boolean
|
|
119
|
+
}
|
|
120
|
+
}
|