osborn 0.9.128 → 0.9.129

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (2) hide show
  1. package/dist/claude-llm.js +14 -6
  2. package/package.json +1 -1
@@ -483,7 +483,12 @@ export class ClaudeLLM extends llm.LLM {
483
483
  return 'claude.agent-sdk';
484
484
  }
485
485
  get model() {
486
- return this.#opts.model || 'claude-opus-4-8'; // Opus 4.8 orchestrator with named sub-agents
486
+ // The [1m] suffix opts into Opus 4.8's 1M context window (Claude Code's
487
+ // context-1m-2025-08-07 beta). WITHOUT it the SDK runs opus at its 200k
488
+ // base, so auto-compaction fired at ~153k every ~10 min (confirmed in the
489
+ // live agent log: compact_boundary pre_tokens≈153k). With [1m] the window
490
+ // is 1M and compaction happens far later. Overridable via opts.model.
491
+ return this.#opts.model || 'claude-opus-4-8[1m]';
487
492
  }
488
493
  get sessionId() {
489
494
  return this.#sessionId;
@@ -766,11 +771,14 @@ export class ClaudeLLM extends llm.LLM {
766
771
  pushMessage(userText, sdkOptions, callbacks) {
767
772
  // Auto-compact threshold — fires PreCompact when context fills to this %.
768
773
  // Higher = uses more of the window before compacting (more context retained
769
- // per turn, but less headroom for the next reply). Raised 85→92 because
770
- // users reported compaction firing too often; 92% pushes it as late as is
771
- // safe before Claude starts hard-capping replies near the 1M limit. Pairs
772
- // with autoCompactWindow=1_000_000 (settings.json, read via settingSources).
774
+ // per turn, but less headroom for the next reply). 92% pushes it as late as
775
+ // is safe before Claude starts hard-capping replies near the 1M limit.
773
776
  process.env.CLAUDE_AUTOCOMPACT_PCT_OVERRIDE = '92';
777
+ // Compaction WINDOW in thousands of tokens (min(actual_context, this)). Now
778
+ // that the model runs with [1m] (1M context), set this to 1000 (=1,000,000)
779
+ // so the compaction budget matches the real window instead of being clamped
780
+ // to opus' 200k base — the actual cause of the ~153k early compaction.
781
+ process.env.CLAUDE_CODE_AUTO_COMPACT_WINDOW = '1000';
774
782
  const userMessage = {
775
783
  type: 'user',
776
784
  message: { role: 'user', content: [{ type: 'text', text: userText }] },
@@ -1004,7 +1012,7 @@ class ClaudeLLMStream extends llm.LLMStream {
1004
1012
  permissionMode: this.#opts.permissionMode,
1005
1013
  allowedTools,
1006
1014
  // model: this.#opts.model || 'haiku', // haiku for speed with limited tools, sonnet for full research capabilities (including tool use trace in response)
1007
- model: this.#opts.model || 'claude-opus-4-8', // Opus 4.8 orchestrator with named sub-agents (Haiku tested but ignored delegation rules)
1015
+ model: this.#opts.model || 'claude-opus-4-8[1m]', // Opus 4.8 + [1m] → 1M context (see get model() note); prevents ~153k early compaction
1008
1016
  enableFileCheckpointing: true,
1009
1017
  settingSources: ['project', 'user'],
1010
1018
  extraArgs: { 'replay-user-messages': null },
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "osborn",
3
- "version": "0.9.128",
3
+ "version": "0.9.129",
4
4
  "description": "Voice AI coding assistant - local agent that connects to Osborn frontend",
5
5
  "type": "module",
6
6
  "bin": {