osborn 0.9.128 → 0.9.129
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/claude-llm.js +14 -6
- package/package.json +1 -1
package/dist/claude-llm.js
CHANGED
|
@@ -483,7 +483,12 @@ export class ClaudeLLM extends llm.LLM {
|
|
|
483
483
|
return 'claude.agent-sdk';
|
|
484
484
|
}
|
|
485
485
|
get model() {
|
|
486
|
-
|
|
486
|
+
// The [1m] suffix opts into Opus 4.8's 1M context window (Claude Code's
|
|
487
|
+
// context-1m-2025-08-07 beta). WITHOUT it the SDK runs opus at its 200k
|
|
488
|
+
// base, so auto-compaction fired at ~153k every ~10 min (confirmed in the
|
|
489
|
+
// live agent log: compact_boundary pre_tokens≈153k). With [1m] the window
|
|
490
|
+
// is 1M and compaction happens far later. Overridable via opts.model.
|
|
491
|
+
return this.#opts.model || 'claude-opus-4-8[1m]';
|
|
487
492
|
}
|
|
488
493
|
get sessionId() {
|
|
489
494
|
return this.#sessionId;
|
|
@@ -766,11 +771,14 @@ export class ClaudeLLM extends llm.LLM {
|
|
|
766
771
|
pushMessage(userText, sdkOptions, callbacks) {
|
|
767
772
|
// Auto-compact threshold — fires PreCompact when context fills to this %.
|
|
768
773
|
// Higher = uses more of the window before compacting (more context retained
|
|
769
|
-
// per turn, but less headroom for the next reply).
|
|
770
|
-
//
|
|
771
|
-
// safe before Claude starts hard-capping replies near the 1M limit. Pairs
|
|
772
|
-
// with autoCompactWindow=1_000_000 (settings.json, read via settingSources).
|
|
774
|
+
// per turn, but less headroom for the next reply). 92% pushes it as late as
|
|
775
|
+
// is safe before Claude starts hard-capping replies near the 1M limit.
|
|
773
776
|
process.env.CLAUDE_AUTOCOMPACT_PCT_OVERRIDE = '92';
|
|
777
|
+
// Compaction WINDOW in thousands of tokens (min(actual_context, this)). Now
|
|
778
|
+
// that the model runs with [1m] (1M context), set this to 1000 (=1,000,000)
|
|
779
|
+
// so the compaction budget matches the real window instead of being clamped
|
|
780
|
+
// to opus' 200k base — the actual cause of the ~153k early compaction.
|
|
781
|
+
process.env.CLAUDE_CODE_AUTO_COMPACT_WINDOW = '1000';
|
|
774
782
|
const userMessage = {
|
|
775
783
|
type: 'user',
|
|
776
784
|
message: { role: 'user', content: [{ type: 'text', text: userText }] },
|
|
@@ -1004,7 +1012,7 @@ class ClaudeLLMStream extends llm.LLMStream {
|
|
|
1004
1012
|
permissionMode: this.#opts.permissionMode,
|
|
1005
1013
|
allowedTools,
|
|
1006
1014
|
// model: this.#opts.model || 'haiku', // haiku for speed with limited tools, sonnet for full research capabilities (including tool use trace in response)
|
|
1007
|
-
model: this.#opts.model || 'claude-opus-4-8', // Opus 4.8
|
|
1015
|
+
model: this.#opts.model || 'claude-opus-4-8[1m]', // Opus 4.8 + [1m] → 1M context (see get model() note); prevents ~153k early compaction
|
|
1008
1016
|
enableFileCheckpointing: true,
|
|
1009
1017
|
settingSources: ['project', 'user'],
|
|
1010
1018
|
extraArgs: { 'replay-user-messages': null },
|