osborn 0.9.129 → 0.9.131
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/claude-llm.js +13 -10
- package/dist/index.js +10 -3
- package/dist/pipeline-direct-llm.d.ts +2 -0
- package/dist/pipeline-direct-llm.js +4 -1
- package/package.json +1 -1
package/dist/claude-llm.js
CHANGED
|
@@ -149,6 +149,16 @@ function ensureCompactionSettings() {
|
|
|
149
149
|
}
|
|
150
150
|
}
|
|
151
151
|
ensureCompactionSettings();
|
|
152
|
+
// Enable Opus 4.8's native 1M context window + set the auto-compact threshold
|
|
153
|
+
// at MODULE LOAD — BEFORE any query() spawns — so they actually apply to the
|
|
154
|
+
// persistent session. Setting these per-message in pushMessage() is too late:
|
|
155
|
+
// the SDK subprocess reads env once at cold-start, so later writes never reach
|
|
156
|
+
// the running query — the reason auto-compaction kept firing at ~150k every
|
|
157
|
+
// turn. ENABLE_1M_CONTEXT is Claude Code's documented switch for the context-1m
|
|
158
|
+
// (1M) window; without it opus runs at its 200K default. NOTE: 1M activation
|
|
159
|
+
// also depends on account entitlement (auto on Team seats; else usage credits).
|
|
160
|
+
process.env.ENABLE_1M_CONTEXT = '1';
|
|
161
|
+
process.env.CLAUDE_AUTOCOMPACT_PCT_OVERRIDE = '92';
|
|
152
162
|
// Research mode tools — full research capabilities
|
|
153
163
|
// Named sub-agents — the orchestrator delegates to these specialists. Each has
|
|
154
164
|
// a specific role, model, and tool set. Module-level + exported so the HTTP
|
|
@@ -769,16 +779,9 @@ export class ClaudeLLM extends llm.LLM {
|
|
|
769
779
|
* @param callbacks - Event callbacks for the background consumer
|
|
770
780
|
*/
|
|
771
781
|
pushMessage(userText, sdkOptions, callbacks) {
|
|
772
|
-
//
|
|
773
|
-
//
|
|
774
|
-
// per
|
|
775
|
-
// is safe before Claude starts hard-capping replies near the 1M limit.
|
|
776
|
-
process.env.CLAUDE_AUTOCOMPACT_PCT_OVERRIDE = '92';
|
|
777
|
-
// Compaction WINDOW in thousands of tokens (min(actual_context, this)). Now
|
|
778
|
-
// that the model runs with [1m] (1M context), set this to 1000 (=1,000,000)
|
|
779
|
-
// so the compaction budget matches the real window instead of being clamped
|
|
780
|
-
// to opus' 200k base — the actual cause of the ~153k early compaction.
|
|
781
|
-
process.env.CLAUDE_CODE_AUTO_COMPACT_WINDOW = '1000';
|
|
782
|
+
// (Compaction/1M env is set once at module load — see ENABLE_1M_CONTEXT /
|
|
783
|
+
// CLAUDE_AUTOCOMPACT_PCT_OVERRIDE near ensureCompactionSettings(). Setting it
|
|
784
|
+
// here per-message was too late for the persistent query.)
|
|
782
785
|
const userMessage = {
|
|
783
786
|
type: 'user',
|
|
784
787
|
message: { role: 'user', content: [{ type: 'text', text: userText }] },
|
package/dist/index.js
CHANGED
|
@@ -2214,7 +2214,7 @@ async function main() {
|
|
|
2214
2214
|
const { readSessionHistory } = await import('./session-access.js');
|
|
2215
2215
|
const history = readSessionHistory(sessionId, workingDir, {
|
|
2216
2216
|
lastN: 10,
|
|
2217
|
-
types: ['assistant'],
|
|
2217
|
+
types: ['assistant', 'tool_use'], // tool_use entries also carry Claude's spoken text
|
|
2218
2218
|
});
|
|
2219
2219
|
recentMessages = history
|
|
2220
2220
|
.filter((m) => m.text)
|
|
@@ -2230,8 +2230,14 @@ async function main() {
|
|
|
2230
2230
|
// BEFORE the previous TTS completed, and we may have already suppressed in-flight
|
|
2231
2231
|
// tts_say events that arrived during that overlap).
|
|
2232
2232
|
const carriedSuppressed = lastInterruption?.suppressedText ?? '';
|
|
2233
|
-
lastInterruption = {
|
|
2234
|
-
|
|
2233
|
+
lastInterruption = {
|
|
2234
|
+
spokenText: fullText,
|
|
2235
|
+
recentMessages,
|
|
2236
|
+
suppressedText: carriedSuppressed,
|
|
2237
|
+
fullBlockText: fullBlockText && fullBlockText !== fullText ? fullBlockText : undefined,
|
|
2238
|
+
timestamp: Date.now(),
|
|
2239
|
+
};
|
|
2240
|
+
console.log(`📋 Interruption context stored (text: ${fullText.length} chars, JSONL: ${recentMessages.length} chars, block: ${fullBlockText?.length ?? 0} chars, suppressed carried: ${carriedSuppressed.length} chars)`);
|
|
2235
2241
|
}
|
|
2236
2242
|
/**
|
|
2237
2243
|
* Append text the agent tried to say while the user was speaking, but which we
|
|
@@ -2271,6 +2277,7 @@ async function main() {
|
|
|
2271
2277
|
spokenText: lastInterruption.spokenText,
|
|
2272
2278
|
recentMessages: lastInterruption.recentMessages,
|
|
2273
2279
|
suppressedText: lastInterruption.suppressedText,
|
|
2280
|
+
fullBlockText: lastInterruption.fullBlockText,
|
|
2274
2281
|
};
|
|
2275
2282
|
lastInterruption = null;
|
|
2276
2283
|
return ctx;
|
|
@@ -21,6 +21,8 @@ export interface InterruptionContext {
|
|
|
21
21
|
* and can re-articulate the relevant bits in its next response.
|
|
22
22
|
*/
|
|
23
23
|
suppressedText: string;
|
|
24
|
+
/** Full text of the interrupted TTS block — fallback when JSONL read returns empty */
|
|
25
|
+
fullBlockText?: string;
|
|
24
26
|
}
|
|
25
27
|
export interface PipelineDirectOptions extends ClaudeLLMOptions {
|
|
26
28
|
onFastBrainResult?: (result: FastBrainPanelResult) => void;
|
|
@@ -122,7 +122,10 @@ export class PipelineDirectLLM extends llm.LLM {
|
|
|
122
122
|
`Anything in "Your recent messages" below that appears AFTER the quoted heard text is content the user did not hear. The user has no memory of it.`,
|
|
123
123
|
``,
|
|
124
124
|
`Your recent messages (full untruncated — you wrote these):`,
|
|
125
|
-
interruptCtx.recentMessages
|
|
125
|
+
interruptCtx.recentMessages
|
|
126
|
+
|| (interruptCtx.fullBlockText
|
|
127
|
+
? `(JSONL unavailable — full TTS block text follows)\n"${interruptCtx.fullBlockText}"`
|
|
128
|
+
: '(no recent messages found)'),
|
|
126
129
|
suppressedBlock,
|
|
127
130
|
``,
|
|
128
131
|
`User's message: "${userText}"`,
|