@orxataguy/tyr 1.0.33 → 1.0.35

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@orxataguy/tyr",
3
- "version": "1.0.33",
3
+ "version": "1.0.35",
4
4
  "type": "module",
5
5
  "bin": {
6
6
  "tyr": "./bin/tyr.js"
@@ -43,9 +43,10 @@ export interface AITool {
43
43
 
44
44
  /**
45
45
  * Nivel de "esfuerzo de razonamiento" (extended thinking) pedido al modelo, independiente del
46
- * vendor concreto. Se traduce a `thinking.budget_tokens` en Anthropic y a `reasoning_effort` en
47
- * los modelos de OpenAI que lo soportan (serie 'o'); Gemini no tiene un equivalente público en
48
- * este momento y el valor se ignora silenciosamente para ese vendor.
46
+ * vendor concreto. Se traduce a `thinking.budget_tokens` en Anthropic, a
47
+ * `generationConfig.thinkingConfig.thinkingBudget` en Gemini 2.5+ (mismo concepto de presupuesto
48
+ * de tokens que Anthropic) y a `reasoning_effort` en los modelos de OpenAI que lo soportan (serie
49
+ * 'o'), que solo acepta un vocabulario fijo (low/medium/high) en lugar de un número de tokens.
49
50
  */
50
51
  export type ThinkingEffort = 'none' | 'low' | 'medium' | 'high' | 'max';
51
52
 
@@ -130,11 +131,6 @@ const RETRY_BASE_DELAY_MS = 500;
130
131
  // touching code. 'balanced' matches DEFAULT_MODELS on purpose — it's the vendor's normal default.
131
132
  type ModelTier = 'cheap' | 'balanced' | 'flagship';
132
133
 
133
- const MODEL_TIERS: Record<AIVendor, Record<ModelTier, string>> = {
134
- anthropic: { cheap: 'claude-haiku-4-5', balanced: 'claude-sonnet-5', flagship: 'claude-opus-4-1' },
135
- openai: { cheap: 'gpt-4o-mini', balanced: 'gpt-4o-mini', flagship: 'gpt-4o' },
136
- gemini: { cheap: 'gemini-2.5-flash', balanced: 'gemini-2.5-pro', flagship: 'gemini-2.5-pro' },
137
- };
138
134
 
139
135
  const MODEL_TIER_ENV: Record<AIVendor, Record<ModelTier, string>> = {
140
136
  anthropic: { cheap: 'ANTHROPIC_AI_MODEL_CHEAP', balanced: 'ANTHROPIC_AI_MODEL_BALANCED', flagship: 'ANTHROPIC_AI_MODEL_FLAGSHIP' },
@@ -142,8 +138,36 @@ const MODEL_TIER_ENV: Record<AIVendor, Record<ModelTier, string>> = {
142
138
  gemini: { cheap: 'GEMINI_AI_MODEL_CHEAP', balanced: 'GEMINI_AI_MODEL_BALANCED', flagship: 'GEMINI_AI_MODEL_FLAGSHIP' },
143
139
  };
144
140
 
145
- /** budget_tokens sent to Anthropic's `thinking` param per effort level. 'none' disables thinking. */
146
- const THINKING_BUDGETS: Record<ThinkingEffort, number> = {
141
+
142
+ const MODEL_TIER_DEFAULT: Record<string,string> = {
143
+ anthropic: 'claude-haiku-4-5-20251001',
144
+ openai: 'gpt-4o-mini-2024-07-18',
145
+ gemini: 'gemini-2.5-flash-lite',
146
+ };
147
+
148
+ /** budget_tokens sent to Anthropic's `thinking` param per effort level. 'none' disables thinking.
149
+ * Each is overridable via env var (see ANTHROPIC_THINKING_BUDGET_ENV); these are the fallback
150
+ * values used when the corresponding env var is not set. */
151
+ const ANTHROPIC_THINKING_BUDGETS: Record<ThinkingEffort, number> = {
152
+ none: 0,
153
+ low: 1024,
154
+ medium: 4096,
155
+ high: 8192,
156
+ max: 16384,
157
+ };
158
+
159
+ const ANTHROPIC_THINKING_BUDGET_ENV: Record<ThinkingEffort, string> = {
160
+ none: 'ANTHROPIC_THINKING_BUDGET_NONE',
161
+ low: 'ANTHROPIC_THINKING_BUDGET_LOW',
162
+ medium: 'ANTHROPIC_THINKING_BUDGET_MEDIUM',
163
+ high: 'ANTHROPIC_THINKING_BUDGET_HIGH',
164
+ max: 'ANTHROPIC_THINKING_BUDGET_MAX',
165
+ };
166
+
167
+ /** Gemini 2.5+'s `generationConfig.thinkingConfig.thinkingBudget` is the same token-budget concept
168
+ * as Anthropic's, so it reuses the same default magnitudes — overridable independently via its
169
+ * own env vars since Gemini's actual valid range differs by model (flash allows 0, pro doesn't). */
170
+ const GEMINI_THINKING_BUDGETS: Record<ThinkingEffort, number> = {
147
171
  none: 0,
148
172
  low: 1024,
149
173
  medium: 4096,
@@ -151,20 +175,92 @@ const THINKING_BUDGETS: Record<ThinkingEffort, number> = {
151
175
  max: 16384,
152
176
  };
153
177
 
178
+ const GEMINI_THINKING_BUDGET_ENV: Record<ThinkingEffort, string> = {
179
+ none: 'GEMINI_THINKING_BUDGET_NONE',
180
+ low: 'GEMINI_THINKING_BUDGET_LOW',
181
+ medium: 'GEMINI_THINKING_BUDGET_MEDIUM',
182
+ high: 'GEMINI_THINKING_BUDGET_HIGH',
183
+ max: 'GEMINI_THINKING_BUDGET_MAX',
184
+ };
185
+
186
+ function resolveThinkingBudget(
187
+ effort: ThinkingEffort,
188
+ defaults: Record<ThinkingEffort, number>,
189
+ env: Record<ThinkingEffort, string>
190
+ ): number {
191
+ return getEnvInt(env[effort], defaults[effort]);
192
+ }
193
+
194
+ /** OpenAI's reasoning models (serie 'o', gpt-5*) solo aceptan un vocabulario fijo para
195
+ * `reasoning_effort` — no un número de tokens — así que aquí se mapea ThinkingEffort a esas
196
+ * cadenas en lugar de a un budget. 'none' está ausente a propósito: esos modelos no permiten
197
+ * desactivar el razonamiento, así que ThinkingEffort 'none' simplemente omite el parámetro
198
+ * (ver openaiReasoningFields). Cada nivel es overridable vía env var por si OpenAI cambia el
199
+ * vocabulario aceptado (p.ej. añade 'minimal') sin tener que tocar código. */
200
+ const OPENAI_REASONING_EFFORTS: Partial<Record<ThinkingEffort, string>> = {
201
+ low: 'low',
202
+ medium: 'medium',
203
+ high: 'high',
204
+ max: 'high',
205
+ };
206
+
207
+ const OPENAI_REASONING_EFFORT_ENV: Partial<Record<ThinkingEffort, string>> = {
208
+ low: 'OPENAI_REASONING_EFFORT_LOW',
209
+ medium: 'OPENAI_REASONING_EFFORT_MEDIUM',
210
+ high: 'OPENAI_REASONING_EFFORT_HIGH',
211
+ max: 'OPENAI_REASONING_EFFORT_MAX',
212
+ };
213
+
214
+ function resolveOpenAIReasoningEffort(effort: ThinkingEffort): string | undefined {
215
+ const envVar = OPENAI_REASONING_EFFORT_ENV[effort];
216
+ const override = envVar ? getEnvString(envVar) : undefined;
217
+ return override ?? OPENAI_REASONING_EFFORTS[effort];
218
+ }
219
+
154
220
  /**
155
221
  * Task-priority → (model tier, thinking effort) matrix. This is the routing table requested by
156
222
  * the architecture spec: callers pick a priority by *intent* ("this is a trivial file", "this
157
223
  * previous attempt failed"), not by model name, and AIVendorManager decides what that costs.
224
+ * Each entry is overridable via env var (see PRIORITY_ROUTING_ENV); these are the fallback values
225
+ * used when the corresponding env var is unset or holds an invalid tier/thinking value.
158
226
  */
159
227
  const PRIORITY_ROUTING: Record<TaskPriority, { tier: ModelTier; thinking: ThinkingEffort }> = {
160
228
  'baja-prioridad': { tier: 'cheap', thinking: 'none' },
161
- 'baja-media-prioridad': { tier: 'cheap', thinking: 'max' },
229
+ 'baja-media-prioridad': { tier: 'cheap', thinking: 'high' },
162
230
  'media-prioridad': { tier: 'balanced', thinking: 'none' },
163
231
  'media-alta-prioridad': { tier: 'balanced', thinking: 'high' },
164
232
  'alta-prioridad': { tier: 'balanced', thinking: 'max' },
165
233
  'muy-alta-prioridad': { tier: 'flagship', thinking: 'max' },
166
234
  };
167
235
 
236
+ const PRIORITY_ROUTING_ENV: Record<TaskPriority, { tier: string; thinking: string }> = {
237
+ 'baja-prioridad': { tier: 'AI_PRIORITY_BAJA_TIER', thinking: 'AI_PRIORITY_BAJA_THINKING' },
238
+ 'baja-media-prioridad': { tier: 'AI_PRIORITY_BAJA_MEDIA_TIER', thinking: 'AI_PRIORITY_BAJA_MEDIA_THINKING' },
239
+ 'media-prioridad': { tier: 'AI_PRIORITY_MEDIA_TIER', thinking: 'AI_PRIORITY_MEDIA_THINKING' },
240
+ 'media-alta-prioridad': { tier: 'AI_PRIORITY_MEDIA_ALTA_TIER', thinking: 'AI_PRIORITY_MEDIA_ALTA_THINKING' },
241
+ 'alta-prioridad': { tier: 'AI_PRIORITY_ALTA_TIER', thinking: 'AI_PRIORITY_ALTA_THINKING' },
242
+ 'muy-alta-prioridad': { tier: 'AI_PRIORITY_MUY_ALTA_TIER', thinking: 'AI_PRIORITY_MUY_ALTA_THINKING' },
243
+ };
244
+
245
+ const VALID_MODEL_TIERS: ModelTier[] = ['cheap', 'balanced', 'flagship'];
246
+ const VALID_THINKING_EFFORTS: ThinkingEffort[] = ['none', 'low', 'medium', 'high', 'max'];
247
+
248
+ /** Resolves the (tier, thinking) route for a priority, letting env vars override either half of
249
+ * PRIORITY_ROUTING independently. An env var holding anything other than a valid tier/thinking
250
+ * value is ignored in favor of the hardcoded default, rather than passed through to the vendor. */
251
+ function resolvePriorityRoute(priority: TaskPriority): { tier: ModelTier; thinking: ThinkingEffort } {
252
+ const defaults = PRIORITY_ROUTING[priority];
253
+ const envVars = PRIORITY_ROUTING_ENV[priority];
254
+
255
+ const tierOverride = getEnvString(envVars.tier)?.toLowerCase() as ModelTier | undefined;
256
+ const thinkingOverride = getEnvString(envVars.thinking)?.toLowerCase() as ThinkingEffort | undefined;
257
+
258
+ return {
259
+ tier: tierOverride && VALID_MODEL_TIERS.includes(tierOverride) ? tierOverride : defaults.tier,
260
+ thinking: thinkingOverride && VALID_THINKING_EFFORTS.includes(thinkingOverride) ? thinkingOverride : defaults.thinking,
261
+ };
262
+ }
263
+
168
264
  /** Ordered priority ladder, used by bumpPriority() to escalate one level at a time. */
169
265
  const PRIORITY_LADDER: TaskPriority[] = [
170
266
  'baja-prioridad',
@@ -255,12 +351,12 @@ export class AIVendorManager {
255
351
  .toString()
256
352
  .toLowerCase() as AIVendor;
257
353
 
258
- const route = PRIORITY_ROUTING[priority];
259
- if (!route) {
354
+ if (!PRIORITY_ROUTING[priority]) {
260
355
  throw new TyrError(`Unknown task priority: '${priority}'`, null, `Use one of: ${PRIORITY_LADDER.join(', ')}`);
261
356
  }
357
+ const route = resolvePriorityRoute(priority);
262
358
 
263
- const tierModel = getEnvString(MODEL_TIER_ENV[vendor][route.tier]) ?? MODEL_TIERS[vendor][route.tier];
359
+ const tierModel = getEnvString(MODEL_TIER_ENV[vendor][route.tier], MODEL_TIER_DEFAULT[vendor]);
264
360
 
265
361
  return {
266
362
  vendor,
@@ -470,7 +566,7 @@ export class AIVendorManager {
470
566
  /** Anthropic's extended thinking requires temperature === 1 and max_tokens strictly greater
471
567
  * than budget_tokens; both are enforced here so callers never have to remember it. */
472
568
  private anthropicThinkingFields(config: VendorConfig): { thinking?: any; temperature: number; max_tokens: number } {
473
- const budget = THINKING_BUDGETS[config.thinking];
569
+ const budget = resolveThinkingBudget(config.thinking, ANTHROPIC_THINKING_BUDGETS, ANTHROPIC_THINKING_BUDGET_ENV);
474
570
  if (!budget) return { temperature: config.temperature, max_tokens: config.maxTokens };
475
571
 
476
572
  return {
@@ -481,14 +577,23 @@ export class AIVendorManager {
481
577
  }
482
578
 
483
579
  /** OpenAI's `reasoning_effort` only exists on the 'o' reasoning model family and only accepts
484
- * low/medium/high (no 'none'/'max') — mapped from our vendor-agnostic ThinkingEffort. Those
485
- * models also reject a custom `temperature`, so it's omitted whenever reasoning_effort is set. */
580
+ * a fixed vocabulary (no 'none') — mapped from our vendor-agnostic ThinkingEffort via
581
+ * OPENAI_REASONING_EFFORTS/_ENV. Those models also reject a custom `temperature`, so it's
582
+ * omitted whenever reasoning_effort is set. */
486
583
  private openaiReasoningFields(config: VendorConfig): { reasoning_effort?: string; temperature?: number } {
487
584
  const isReasoningModel = /^o\d/.test(config.model);
488
585
  if (!isReasoningModel || config.thinking === 'none') return { temperature: config.temperature };
489
586
 
490
- const effortMap: Partial<Record<ThinkingEffort, string>> = { low: 'low', medium: 'medium', high: 'high', max: 'high' };
491
- return { reasoning_effort: effortMap[config.thinking] };
587
+ return { reasoning_effort: resolveOpenAIReasoningEffort(config.thinking) };
588
+ }
589
+
590
+ /** Gemini 2.5+'s `thinkingConfig.thinkingBudget` is omitted entirely for 'none' rather than
591
+ * sent as 0, because 0 is invalid on Pro models (which can't fully disable thinking) — only
592
+ * Flash accepts it. Omitting the field is the safe "off" behavior across both. */
593
+ private geminiThinkingFields(config: VendorConfig): { thinkingConfig?: { thinkingBudget: number } } {
594
+ if (config.thinking === 'none') return {};
595
+ const budget = resolveThinkingBudget(config.thinking, GEMINI_THINKING_BUDGETS, GEMINI_THINKING_BUDGET_ENV);
596
+ return { thinkingConfig: { thinkingBudget: budget } };
492
597
  }
493
598
 
494
599
  private buildRequest(config: VendorConfig, messages: AIMessage[], stream: boolean, tools?: AITool[]): VendorRequest {
@@ -541,6 +646,7 @@ export class AIVendorManager {
541
646
 
542
647
  case 'gemini': {
543
648
  const action = stream ? 'streamGenerateContent?alt=sse&' : 'generateContent?';
649
+ const thinkingFields = this.geminiThinkingFields(config);
544
650
  return {
545
651
  url: `https://generativelanguage.googleapis.com/v1beta/models/${config.model}:${action}key=${config.apiKey}`,
546
652
  headers: { 'content-type': 'application/json' },
@@ -550,6 +656,7 @@ export class AIVendorManager {
550
656
  generationConfig: {
551
657
  temperature: config.temperature,
552
658
  maxOutputTokens: config.maxTokens,
659
+ ...thinkingFields,
553
660
  },
554
661
  ...(vendorTools ? { tools: vendorTools } : {}),
555
662
  },