@code-yeongyu/senpi-ai 2026.9.29-5 → 2026.9.30

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -323,6 +323,38 @@ function hasOpenAIExtendedPromptCache(model) {
323
323
  }
324
324
  return OPENAI_EXTENDED_CACHE_MODEL_ID.test(model.id);
325
325
  }
326
+ /** Env that puts Claude Code on API-key, gateway or cloud billing, where it caches the main conversation for 5 minutes. */
327
+ const CLAUDE_CODE_API_BILLING_ENV = [
328
+ "ANTHROPIC_API_KEY",
329
+ "ANTHROPIC_AUTH_TOKEN",
330
+ "ANTHROPIC_BASE_URL",
331
+ "CLAUDE_CODE_USE_BEDROCK",
332
+ "CLAUDE_CODE_USE_VERTEX",
333
+ "CLAUDE_CODE_USE_FOUNDRY",
334
+ ];
335
+ function isEnabledFlag(value) {
336
+ return value !== undefined && !/^(?:|0|false|no|off)$/i.test(value.trim());
337
+ }
338
+ /**
339
+ * The Claude SDK lane's cache TTL is chosen by Claude Code, not senpi: `CLAUDE_CODE_PROMPT_CACHE_TTL`
340
+ * (`5m` | `1h`) wins, then `FORCE_PROMPT_CACHING_5M` and `ENABLE_PROMPT_CACHING_1H`; otherwise a Claude
341
+ * subscription gets 1 hour and API-key, gateway or cloud billing gets 5 minutes. A subscription past its
342
+ * usage limits also drops to 5 minutes, which nothing here can observe.
343
+ */
344
+ function claudeCodePromptCacheTtlSeconds(env) {
345
+ const explicit = getProviderEnvValue("CLAUDE_CODE_PROMPT_CACHE_TTL", env)?.trim().toLowerCase();
346
+ if (explicit === "5m")
347
+ return PROMPT_CACHE_TTL_SHORT_SECONDS;
348
+ if (explicit === "1h")
349
+ return PROMPT_CACHE_TTL_LONG_SECONDS;
350
+ if (isEnabledFlag(getProviderEnvValue("FORCE_PROMPT_CACHING_5M", env)))
351
+ return PROMPT_CACHE_TTL_SHORT_SECONDS;
352
+ if (isEnabledFlag(getProviderEnvValue("ENABLE_PROMPT_CACHING_1H", env)))
353
+ return PROMPT_CACHE_TTL_LONG_SECONDS;
354
+ return CLAUDE_CODE_API_BILLING_ENV.some((name) => getProviderEnvValue(name, env) !== undefined)
355
+ ? PROMPT_CACHE_TTL_SHORT_SECONDS
356
+ : PROMPT_CACHE_TTL_LONG_SECONDS;
357
+ }
326
358
  /**
327
359
  * Classify the active model's prompt-cache lifetime from the provider's documented cache contract.
328
360
  * `cacheRetention: "none"` (or `PI_CACHE_RETENTION` resolving to it) always wins.
@@ -330,8 +362,7 @@ function hasOpenAIExtendedPromptCache(model) {
330
362
  export function resolvePromptCacheLifetime(model, env) {
331
363
  switch (model.api) {
332
364
  case "claude-sdk-oauth":
333
- // The Claude SDK owns prompt caching for this lane and uses Anthropic's default 5m TTL.
334
- return ttl(PROMPT_CACHE_TTL_SHORT_SECONDS);
365
+ return ttl(claudeCodePromptCacheTtlSeconds(env));
335
366
  case "anthropic-messages": {
336
367
  const anthropicModel = model;
337
368
  const retention = resolveAnthropicCacheRetention(anthropicModel.cacheRetention, env, "short");
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@code-yeongyu/senpi-ai",
3
- "version": "2026.9.29-5",
3
+ "version": "2026.9.30",
4
4
  "description": "Unified LLM API with automatic model discovery and provider configuration",
5
5
  "type": "module",
6
6
  "main": "./dist/index.js",
@@ -85,7 +85,7 @@
85
85
  "dependencies": {
86
86
  "@anthropic-ai/sdk": "0.127.0",
87
87
  "@aws-sdk/client-bedrock-runtime": "3.1136.0",
88
- "@earendil-works/pi-telemetry": "npm:@code-yeongyu/senpi-telemetry@2026.9.29-5",
88
+ "@earendil-works/pi-telemetry": "npm:@code-yeongyu/senpi-telemetry@2026.9.30",
89
89
  "@google/genai": "2.23.0",
90
90
  "@smithy/node-http-handler": "4.12.1",
91
91
  "http-proxy-agent": "9.1.0",