dsh-autotier 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (60) hide show
  1. package/AGENTS.md +93 -0
  2. package/CHANGELOG.md +85 -0
  3. package/LICENSE +201 -0
  4. package/README.es.md +247 -0
  5. package/README.hi.md +241 -0
  6. package/README.md +245 -0
  7. package/README.pt.md +246 -0
  8. package/README.zh.md +221 -0
  9. package/SECURITY.md +55 -0
  10. package/THIRD_PARTY_NOTICES.md +63 -0
  11. package/cordis.patch.yml +125 -0
  12. package/docs/preset-row.md +61 -0
  13. package/docs/supporting-lanes.md +45 -0
  14. package/lib/index.js +2848 -0
  15. package/lib/types/command.d.ts +17 -0
  16. package/lib/types/command.d.ts.map +1 -0
  17. package/lib/types/config.d.ts +94 -0
  18. package/lib/types/config.d.ts.map +1 -0
  19. package/lib/types/guard-rules.d.ts +97 -0
  20. package/lib/types/guard-rules.d.ts.map +1 -0
  21. package/lib/types/guard.d.ts +70 -0
  22. package/lib/types/guard.d.ts.map +1 -0
  23. package/lib/types/index.d.ts +60 -0
  24. package/lib/types/index.d.ts.map +1 -0
  25. package/lib/types/intent.d.ts +179 -0
  26. package/lib/types/intent.d.ts.map +1 -0
  27. package/lib/types/judge.d.ts +50 -0
  28. package/lib/types/judge.d.ts.map +1 -0
  29. package/lib/types/policy.d.ts +109 -0
  30. package/lib/types/policy.d.ts.map +1 -0
  31. package/lib/types/routing.d.ts +135 -0
  32. package/lib/types/routing.d.ts.map +1 -0
  33. package/lib/types/schema.d.ts +134 -0
  34. package/lib/types/schema.d.ts.map +1 -0
  35. package/lib/types/service.d.ts +67 -0
  36. package/lib/types/service.d.ts.map +1 -0
  37. package/lib/types/state.d.ts +46 -0
  38. package/lib/types/state.d.ts.map +1 -0
  39. package/lib/types/tiers.d.ts +103 -0
  40. package/lib/types/tiers.d.ts.map +1 -0
  41. package/lib/types/tools.d.ts +26 -0
  42. package/lib/types/tools.d.ts.map +1 -0
  43. package/lib/types/types.d.ts +96 -0
  44. package/lib/types/types.d.ts.map +1 -0
  45. package/package.json +179 -0
  46. package/src/command.ts +73 -0
  47. package/src/config.ts +358 -0
  48. package/src/guard-rules.ts +303 -0
  49. package/src/guard.ts +285 -0
  50. package/src/index.ts +149 -0
  51. package/src/intent.ts +484 -0
  52. package/src/judge.ts +150 -0
  53. package/src/policy.ts +246 -0
  54. package/src/routing.ts +575 -0
  55. package/src/schema.ts +295 -0
  56. package/src/service.ts +131 -0
  57. package/src/state.ts +134 -0
  58. package/src/tiers.ts +212 -0
  59. package/src/tools.ts +128 -0
  60. package/src/types.ts +120 -0
package/src/tiers.ts ADDED
@@ -0,0 +1,212 @@
1
+ /**
2
+ * Pure tier arithmetic: the adapter-owned effort ladder, tier-route
3
+ * application, the effort-first escalation ladder, and the fallback-chain
4
+ * classifier/advance helpers.
5
+ *
6
+ * The fallback vocabulary is ported from `dsh-tier-router`'s `lib/pure.js`
7
+ * (MIT; see `THIRD_PARTY_NOTICES.md`), with two corrections: the effort ladder
8
+ * is DeepSeek's `off | low | high | max` (upstream's `medium` does not exist and
9
+ * would fail every request with `UNSUPPORTED_REASONING_EFFORT`), and the
10
+ * classification is split into permanent/transient so the request-error handler
11
+ * can honour the division of labour with `dsh-llm-retry` (permanent codes switch
12
+ * the chain immediately; transient codes wait for retry exhaustion).
13
+ *
14
+ * @module dsh-autotier/tiers
15
+ */
16
+
17
+ import type { LlmCallConfig } from '@deepseek-ai/dsh-llm'
18
+ import type { EffortId, TierId, TierRoute } from './types.ts'
19
+
20
+ /** The adapter-owned effort ladder, cheapest to strongest. */
21
+ export const EFFORT_LADDER = ['off', 'low', 'high', 'max'] as const satisfies readonly EffortId[]
22
+
23
+ /** Position of one effort on the ladder (0..3); unknown ids rank lowest. */
24
+ export function effortRank(effort: string): number {
25
+ const index = (EFFORT_LADDER as readonly string[]).indexOf(effort)
26
+ return index === -1 ? 0 : index
27
+ }
28
+
29
+ /**
30
+ * The next effort step above `current`, never above `ceiling`.
31
+ * @param current - the effort currently in force.
32
+ * @param ceiling - the strongest effort to consider (default `max`).
33
+ * @returns the next effort id, or null when already at or above the ceiling.
34
+ */
35
+ export function nextEffortStep(current: EffortId, ceiling: EffortId = 'max'): EffortId | null {
36
+ const start = effortRank(current)
37
+ const stop = effortRank(ceiling)
38
+ if (start >= stop) return null
39
+ return EFFORT_LADDER[start + 1] ?? null
40
+ }
41
+
42
+ /** Whether two routes land on the same provider/model/effort triple. */
43
+ export function routeEquals(a: TierRoute, b: TierRoute): boolean {
44
+ return a.provider === b.provider && a.model === b.model && a.effort === b.effort
45
+ }
46
+
47
+ /** One adapter-owned effort id as the LLM call config expects it (branded at the seam). */
48
+ function brandedEffort(effort: string): NonNullable<LlmCallConfig['reasoningEffort']> {
49
+ // The adapter's `ReasoningEffortId` is a branded string; the brand is opaque
50
+ // by design, and our vocabulary is validated against it before it gets here.
51
+ return effort as unknown as NonNullable<LlmCallConfig['reasoningEffort']>
52
+ }
53
+
54
+ /**
55
+ * Apply one tier route to a request configuration. The sampling scalars the
56
+ * session already chose (`temperature`, `maxTokens`, `stop`) are preserved
57
+ * exactly; the provider/model/effort triple is replaced. Returns the input
58
+ * object unchanged when the route already matches, so the caller can skip a
59
+ * logged header change.
60
+ *
61
+ * @param base - the configuration the loop proposed.
62
+ * @param target - the tier landing to apply.
63
+ * @returns the replacement configuration (or `base` when identical).
64
+ */
65
+ export function resolveRoute(base: LlmCallConfig, target: TierRoute): LlmCallConfig {
66
+ const sameLanding = base.provider === target.provider && base.model === target.model
67
+ if (sameLanding && (target.effort === undefined || base.reasoningEffort === target.effort)) return base
68
+ const next: LlmCallConfig = { provider: target.provider, model: target.model }
69
+ // An absent target effort means "follow the session": carry the effort the
70
+ // request already carries instead of dropping it (a dropped effort silently
71
+ // falls back to the adapter default, which is the strongest one).
72
+ const effort = target.effort ?? base.reasoningEffort
73
+ if (effort !== undefined) next.reasoningEffort = brandedEffort(effort)
74
+ if (base.temperature !== undefined) next.temperature = base.temperature
75
+ if (base.maxTokens !== undefined) next.maxTokens = base.maxTokens
76
+ if (base.stop !== undefined) next.stop = base.stop
77
+ return next
78
+ }
79
+
80
+ /** One rung of the effort-first escalation ladder. */
81
+ export interface EscalationRung {
82
+ /** The landing this rung applies. */
83
+ readonly route: TierRoute
84
+ /** The tier the rung belongs to (an effort rung on the cheap model is still cheap). */
85
+ readonly tier: TierId
86
+ /** Why this rung exists, for logs and `/tier status`. */
87
+ readonly note: string
88
+ }
89
+
90
+ /**
91
+ * Build the effort-first escalation ladder: raise the current model's effort
92
+ * one step at a time (the KV prefix survives, and the official model-selection
93
+ * notice is not emitted for an effort-only change) before paying for a model
94
+ * switch. When both tiers share one model the ladder collapses to a single
95
+ * effort-only rung.
96
+ *
97
+ * @param base - the configuration the loop proposed for the failing step.
98
+ * @param cheap - the cheap tier landing.
99
+ * @param strong - the strong tier landing.
100
+ * @returns the ordered rungs; empty when escalation cannot change anything.
101
+ */
102
+ export function escalationLadder(base: LlmCallConfig, cheap: TierRoute, strong: TierRoute): EscalationRung[] {
103
+ const rungs: EscalationRung[] = []
104
+ const baseEffort = (base.reasoningEffort ?? cheap.effort ?? 'low') as EffortId
105
+ const onCheapModel = base.provider === cheap.provider && base.model === cheap.model
106
+ const sharedModel = strong.provider === cheap.provider && strong.model === cheap.model
107
+ if (sharedModel) {
108
+ if (strong.effort !== undefined && strong.effort !== baseEffort) {
109
+ rungs.push({
110
+ route: { provider: strong.provider, model: strong.model, effort: strong.effort },
111
+ tier: 'strong',
112
+ note: `effort-only escalation on the shared model (${baseEffort} -> ${strong.effort})`,
113
+ })
114
+ }
115
+ return rungs
116
+ }
117
+ if (onCheapModel) {
118
+ let step = nextEffortStep(baseEffort)
119
+ while (step !== null) {
120
+ rungs.push({
121
+ route: { provider: cheap.provider, model: cheap.model, effort: step },
122
+ tier: 'cheap',
123
+ note: `cheap-tier effort rung ${step}`,
124
+ })
125
+ step = nextEffortStep(step)
126
+ }
127
+ }
128
+ rungs.push({
129
+ route: strong.effort === undefined
130
+ ? { provider: strong.provider, model: strong.model }
131
+ : { provider: strong.provider, model: strong.model, effort: strong.effort },
132
+ tier: 'strong',
133
+ note: 'switch to the strong model',
134
+ })
135
+ return rungs
136
+ }
137
+
138
+ /** Failure codes that mean the route itself is unusable: switch the chain now. */
139
+ export const FALLBACK_PERMANENT_CODES = [
140
+ 'UNKNOWN_MODEL',
141
+ 'MISSING_CREDENTIAL',
142
+ 'INVALID_CREDENTIAL',
143
+ 'QUOTA',
144
+ ] as const
145
+
146
+ /** Failure codes owned by `dsh-llm-retry` first: switch the chain only after retries are exhausted. */
147
+ export const FALLBACK_TRANSIENT_CODES = ['RATE_LIMIT', 'SERVER', 'TIMEOUT', 'TRANSPORT'] as const
148
+
149
+ /** Failure codes that must never switch the model. */
150
+ export const FALLBACK_IGNORE_CODES = [
151
+ 'CONTEXT_WINDOW_EXCEEDED',
152
+ 'UNSUPPORTED_REASONING_EFFORT',
153
+ 'ABORTED',
154
+ 'EMPTY_RESPONSE',
155
+ ] as const
156
+
157
+ /** How one failure relates to the fallback chain. */
158
+ export type FallbackClass = 'permanent' | 'transient' | 'ignore' | 'unknown'
159
+
160
+ /**
161
+ * Classify one failed model request for the fallback machinery.
162
+ * @param failure - the normalized failure facts (`code` and optional `status`).
163
+ * @returns the chain verdict.
164
+ */
165
+ export function classifyFallback(failure: { code?: unknown; status?: unknown } | undefined): FallbackClass {
166
+ if (failure === undefined || failure === null || typeof failure !== 'object') return 'unknown'
167
+ const code = typeof failure.code === 'string' ? failure.code : undefined
168
+ if (code !== undefined) {
169
+ if ((FALLBACK_PERMANENT_CODES as readonly string[]).includes(code)) return 'permanent'
170
+ if ((FALLBACK_TRANSIENT_CODES as readonly string[]).includes(code)) return 'transient'
171
+ if ((FALLBACK_IGNORE_CODES as readonly string[]).includes(code)) return 'ignore'
172
+ }
173
+ if (typeof failure.status === 'number' && failure.status >= 500) return 'transient'
174
+ return 'unknown'
175
+ }
176
+
177
+ /** One agent's position in one tier's fallback chain. */
178
+ export interface FallbackRecord {
179
+ /** The tier whose chain this record belongs to. */
180
+ tier: TierId
181
+ /** Index in that tier's chain; -1 means the tier's own landing. */
182
+ index: number
183
+ /** Epoch millis until which the record stays in force. */
184
+ until: number
185
+ }
186
+
187
+ /** Whether a fallback record is currently pinning the agent to a chain entry. */
188
+ export function fallbackActive(record: FallbackRecord | undefined, now: number): boolean {
189
+ return record !== undefined && record.index >= 0 && record.until > now
190
+ }
191
+
192
+ /**
193
+ * Advance a fallback record one step down one tier's chain.
194
+ * @param record - the current record (absent or from another tier = start of this chain).
195
+ * @param tier - the tier whose chain is being walked.
196
+ * @param chainLength - the number of configured fallback entries.
197
+ * @param now - current epoch millis.
198
+ * @param ttlMs - how long the new entry stays in force.
199
+ * @returns the next record, or null when the chain is exhausted.
200
+ */
201
+ export function advanceFallback(
202
+ record: FallbackRecord | undefined,
203
+ tier: TierId,
204
+ chainLength: number,
205
+ now: number,
206
+ ttlMs: number,
207
+ ): FallbackRecord | null {
208
+ const index = record !== undefined && record.tier === tier ? record.index : -1
209
+ const next = index + 1
210
+ if (next >= chainLength) return null
211
+ return { tier, index: next, until: now + (Number.isFinite(ttlMs) && ttlMs > 0 ? ttlMs : 300_000) }
212
+ }
package/src/tools.ts ADDED
@@ -0,0 +1,128 @@
1
+ /**
2
+ * The two read-only tools: `tier_status` reports the live routing state, and
3
+ * `tier_route` dry-runs the classifier on one intent string without sending a
4
+ * model request.
5
+ * @module dsh-autotier/tools
6
+ */
7
+
8
+ import type { Context } from '@deepseek-ai/cordis'
9
+ import type { Agent } from '@deepseek-ai/dsh-agent'
10
+ import { defineTool, type ToolDefinition } from '@deepseek-ai/dsh-tools'
11
+ import { classifyIntent } from './intent.ts'
12
+ import { escalationActive } from './policy.ts'
13
+ import type { AutotierService } from './service.ts'
14
+ import type { AgentStateStore } from './state.ts'
15
+
16
+ /** Services the tools read. */
17
+ export interface ToolServices {
18
+ readonly service: AutotierService
19
+ readonly states: AgentStateStore
20
+ }
21
+
22
+ /** Render one canonical value as a single text block. */
23
+ function textBlock(value: unknown): { type: 'text'; text: string }[] {
24
+ return [{ type: 'text', text: typeof value === 'string' ? value : JSON.stringify(value, null, 2) }]
25
+ }
26
+
27
+ /** Build the `tier_status` tool. */
28
+ export function tierStatusTool({ service, states }: ToolServices): ToolDefinition {
29
+ return defineTool({
30
+ name: 'tier_status',
31
+ description:
32
+ 'Report the live dsh-autotier routing state: the configured tier landings, the effective mode for this session, the last intent classification, and whether a failure escalation or fallback is in force. Read-only.',
33
+ parameters: {},
34
+ output: {
35
+ schema: {
36
+ type: 'object',
37
+ additionalProperties: false,
38
+ properties: {
39
+ mode: { type: 'string' },
40
+ strong: { type: 'string' },
41
+ cheap: { type: 'string' },
42
+ vision: { type: 'string' },
43
+ guardEnabled: { type: 'boolean' },
44
+ appliedTier: { type: 'string' },
45
+ lastScenario: { type: 'string' },
46
+ lastConfidence: { type: 'number' },
47
+ escalated: { type: 'boolean' },
48
+ planActive: { type: 'boolean' },
49
+ guardDenials: { type: 'integer' },
50
+ lastDenialRule: { type: 'string' },
51
+ },
52
+ },
53
+ render: (_args, value) => textBlock(value),
54
+ },
55
+ execute: async (_args, exec) => {
56
+ const agent = exec.agent as Agent | undefined
57
+ const state = agent === undefined ? undefined : states.for(agent)
58
+ const status = service.status()
59
+ return {
60
+ mode: state?.override ?? status.mode,
61
+ strong: `${status.tiers.strong.provider}/${status.tiers.strong.model}${status.tiers.strong.effort === undefined ? '' : `@${status.tiers.strong.effort}`}`,
62
+ cheap: `${status.tiers.cheap.provider}/${status.tiers.cheap.model}${status.tiers.cheap.effort === undefined ? '' : `@${status.tiers.cheap.effort}`}`,
63
+ vision: `${status.tiers.vision.provider}/${status.tiers.vision.model}`,
64
+ guardEnabled: status.guard.enabled,
65
+ appliedTier: state?.appliedTier ?? '',
66
+ lastScenario: state?.decision?.scenario ?? '',
67
+ lastConfidence: state?.decision?.confidence ?? 0,
68
+ escalated: state === undefined ? false : escalationActive(state, Date.now()),
69
+ planActive: state?.planActive ?? false,
70
+ guardDenials: state?.denials ?? 0,
71
+ lastDenialRule: state?.lastDenial ?? '',
72
+ }
73
+ },
74
+ })
75
+ }
76
+
77
+ /** Build the `tier_route` tool. */
78
+ export function tierRouteTool({ service }: ToolServices): ToolDefinition {
79
+ return defineTool({
80
+ name: 'tier_route',
81
+ description:
82
+ 'Dry-run the intent classifier on one instruction and report which tier it would use, without sending a model request. Read-only.',
83
+ parameters: {
84
+ text: { type: 'string', description: 'The instruction to classify.', required: true as const },
85
+ },
86
+ output: {
87
+ schema: {
88
+ type: 'object',
89
+ additionalProperties: false,
90
+ properties: {
91
+ tier: { type: 'string' },
92
+ scenario: { type: 'string' },
93
+ confidence: { type: 'number' },
94
+ fingerprint: { type: 'string' },
95
+ reason: { type: 'string' },
96
+ },
97
+ },
98
+ render: (_args, value) => textBlock(value),
99
+ },
100
+ execute: async (args) => {
101
+ const config = service.config()
102
+ const result = classifyIntent({
103
+ text: args.text,
104
+ toolNames: [],
105
+ hasImage: false,
106
+ messageCount: 0,
107
+ cwd: '',
108
+ }, { rules: service.rules(), scenarios: config.intent.scenarios })
109
+ return {
110
+ tier: result.tier,
111
+ scenario: result.scenario,
112
+ confidence: result.confidence,
113
+ fingerprint: result.fingerprint,
114
+ reason: result.reasons.join('; '),
115
+ }
116
+ },
117
+ })
118
+ }
119
+
120
+ /**
121
+ * Register both tools on the plugin fiber.
122
+ * @param ctx - the plugin context (must have `tools`).
123
+ * @param services - the service and the state store.
124
+ */
125
+ export function registerTierTools(ctx: Context, services: ToolServices): void {
126
+ ctx.tools.register(tierStatusTool(services))
127
+ ctx.tools.register(tierRouteTool(services))
128
+ }
package/src/types.ts ADDED
@@ -0,0 +1,120 @@
1
+ /**
2
+ * Shared vocabulary for dsh-autotier: the tier ids, the adapter-owned reasoning
3
+ * effort ids, the routing modes, and the public route/status shapes every
4
+ * module and the `ctx.autotier` service speak.
5
+ * @module dsh-autotier/types
6
+ */
7
+
8
+ /**
9
+ * Service Definition (contract layer): the two cost tiers this plugin routes
10
+ * between. `strong` plans complex intent and reviews high-risk work; `cheap`
11
+ * implements it. The ids are stable wire/settings vocabulary.
12
+ */
13
+ export const TIER_IDS = ['strong', 'cheap'] as const
14
+
15
+ /** One of the two cost tiers. */
16
+ export type TierId = (typeof TIER_IDS)[number]
17
+
18
+ /**
19
+ * Reasoning-effort ids owned by the provider adapter. The DeepSeek adapter
20
+ * accepts exactly these four (`packages/llm/llm-deepseek/src/index.ts:133`);
21
+ * `medium` does not exist, and an unsupported value fails every request with
22
+ * `UNSUPPORTED_REASONING_EFFORT` instead of degrading.
23
+ */
24
+ export const EFFORT_IDS = ['off', 'low', 'high', 'max'] as const
25
+
26
+ /** One adapter-owned reasoning-effort id. */
27
+ export type EffortId = (typeof EFFORT_IDS)[number]
28
+
29
+ /**
30
+ * Routing modes. `auto` is the fully automatic path; the rest are explicit
31
+ * user overrides (`/tier strong|cheap|off`). `delegated` means the session
32
+ * carries an explicit model selection that autotier must not fight.
33
+ */
34
+ export const ROUTING_MODES = ['auto', 'strong', 'cheap', 'delegated', 'off'] as const
35
+
36
+ /** One routing mode. */
37
+ export type RoutingMode = (typeof ROUTING_MODES)[number]
38
+
39
+ /** Intent classes the rule layer and the judge distinguish. */
40
+ export const SCENARIOS = [
41
+ 'coding',
42
+ 'review',
43
+ 'planning',
44
+ 'retrieval',
45
+ 'batch',
46
+ 'daily',
47
+ 'longText',
48
+ 'multimodal',
49
+ ] as const
50
+
51
+ /** One intent class. */
52
+ export type Scenario = (typeof SCENARIOS)[number]
53
+
54
+ /** Cost/quality arbitration direction when the signals are ambiguous. */
55
+ export const COST_MODES = ['cost-first', 'quality-first', 'balanced'] as const
56
+
57
+ /** One cost/quality arbitration direction. */
58
+ export type CostMode = (typeof COST_MODES)[number]
59
+
60
+ /**
61
+ * One concrete tier landing: the provider/model pair plus the reasoning effort
62
+ * the request should carry. `effort` is omitted only when the tier inherits
63
+ * the session's own effort (followSession).
64
+ */
65
+ export interface TierRoute {
66
+ readonly provider: string
67
+ readonly model: string
68
+ readonly effort?: EffortId
69
+ }
70
+
71
+ /**
72
+ * Why a tier was chosen. `source` is the arbitration layer that won, so an
73
+ * operator can tell a rule hit from a judge call from an escalation.
74
+ */
75
+ export type RouteSource =
76
+ | 'manual'
77
+ | 'rule'
78
+ | 'judge'
79
+ | 'posterior'
80
+ | 'plan-mode'
81
+ | 'guard'
82
+ | 'escalation'
83
+ | 'fallback'
84
+ | 'default'
85
+
86
+ /** One routing decision with its provenance. */
87
+ export interface RouteDecision {
88
+ /** The tier the request should use. */
89
+ readonly tier: TierId
90
+ /** The concrete landing. */
91
+ readonly route: TierRoute
92
+ /** The arbitration layer that produced this decision. */
93
+ readonly source: RouteSource
94
+ /** 0..1 confidence of the intent classification (1 for explicit overrides). */
95
+ readonly confidence: number
96
+ /** Short human-readable reason, safe for logs and `/tier status`. */
97
+ readonly reason: string
98
+ }
99
+
100
+ /**
101
+ * Service Definition (contract layer): the read-only status snapshot served by
102
+ * `ctx.autotier.status()` and rendered by `/tier status` and the `tier_status`
103
+ * tool. Third-party plugins may depend on this shape.
104
+ */
105
+ export interface AutotierStatus {
106
+ /** Current routing mode. */
107
+ readonly mode: RoutingMode
108
+ /** Configured landing per tier (effort omitted when inherited). */
109
+ readonly tiers: { readonly strong: TierRoute; readonly cheap: TierRoute; readonly vision: TierRoute }
110
+ /** Guard switches. */
111
+ readonly guard: { readonly enabled: boolean; readonly tiers: readonly TierId[] }
112
+ /** Escalation switches. */
113
+ readonly escalation: {
114
+ readonly threshold: number
115
+ readonly windowMs: number
116
+ readonly ttlMs: number
117
+ readonly fallbackTtlMs: number
118
+ readonly signature: boolean
119
+ }
120
+ }