headlesscode 1.0.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/ATTRIBUTION.md +53 -0
- package/CODE_OF_CONDUCT.md +130 -0
- package/CONTRIBUTING.md +107 -0
- package/LICENSE +202 -0
- package/README.md +486 -0
- package/SECURITY.md +211 -0
- package/bin/headlesscode.mjs +83 -0
- package/package.json +63 -0
- package/shared/prompts/review-mode-prompt-short.md +93 -0
- package/shared/prompts/review-mode-prompt.md +281 -0
- package/shared/rules-code/rules.md +22 -0
- package/shared/stacks/cpp/rules.md +30 -0
- package/shared/stacks/fastapi/rules.md +30 -0
- package/shared/stacks/javascript/rules.md +37 -0
- package/shared/stacks/postgresql/rules.md +31 -0
- package/shared/stacks/python/rules.md +35 -0
- package/shared/stacks/react/rules.md +11 -0
- package/shared/stacks/typescript/rules.md +10 -0
- package/src/budget/budget.ts +221 -0
- package/src/budget/concurrency.ts +126 -0
- package/src/budget/cost.ts +309 -0
- package/src/budget/index.ts +8 -0
- package/src/checkpoints/cli.ts +256 -0
- package/src/checkpoints/service.ts +227 -0
- package/src/cli.ts +1535 -0
- package/src/cloud/docker-provider.ts +334 -0
- package/src/cloud/provider.ts +300 -0
- package/src/codeintel/call-graph.ts +78 -0
- package/src/codeintel/find-references.ts +123 -0
- package/src/codeintel/go-to-definition.ts +193 -0
- package/src/codeintel/handlers.ts +190 -0
- package/src/codeintel/import-graph.ts +173 -0
- package/src/codeintel/outline.ts +180 -0
- package/src/codeintel/position.ts +77 -0
- package/src/codeintel/program.ts +350 -0
- package/src/codeintel/rename-symbol.ts +213 -0
- package/src/codeintel/tools.ts +280 -0
- package/src/codemap/build.ts +135 -0
- package/src/codemap/cli.ts +190 -0
- package/src/codemap/extract.ts +339 -0
- package/src/codemap/files.ts +236 -0
- package/src/codemap/fingerprint.ts +65 -0
- package/src/codemap/flows.ts +62 -0
- package/src/codemap/html.ts +451 -0
- package/src/codemap/lock.ts +80 -0
- package/src/codemap/types.ts +101 -0
- package/src/codesearch/airunner-embedder.ts +185 -0
- package/src/codesearch/chunk.ts +339 -0
- package/src/codesearch/cli.ts +223 -0
- package/src/codesearch/embedder.ts +332 -0
- package/src/codesearch/files.ts +280 -0
- package/src/codesearch/index.ts +469 -0
- package/src/codesearch/ollama-embedder.ts +205 -0
- package/src/codesearch/search.ts +141 -0
- package/src/codesearch/types.ts +100 -0
- package/src/config/mode-models.ts +218 -0
- package/src/dashboard/aggregate.ts +364 -0
- package/src/dashboard/chat-thread.ts +141 -0
- package/src/dashboard/checkpoints.ts +124 -0
- package/src/dashboard/cli.ts +193 -0
- package/src/dashboard/codemap.ts +44 -0
- package/src/dashboard/files.ts +121 -0
- package/src/dashboard/page.ts +2803 -0
- package/src/dashboard/self-improvement-metrics.ts +282 -0
- package/src/dashboard/server.ts +1103 -0
- package/src/dashboard/session-launch.ts +310 -0
- package/src/dashboard/timeline.ts +273 -0
- package/src/dashboard/tool-exec.ts +107 -0
- package/src/dashboard/trend-cli.ts +141 -0
- package/src/dashboard/trend.ts +413 -0
- package/src/decision-proxy/cli.ts +261 -0
- package/src/decision-proxy/proxy.ts +569 -0
- package/src/deploy/gate-cli.ts +147 -0
- package/src/deploy/gate.ts +254 -0
- package/src/engine/condense.ts +512 -0
- package/src/engine/events.ts +428 -0
- package/src/engine/handoff.ts +71 -0
- package/src/engine/lazy-tools.ts +160 -0
- package/src/engine/local-explore.ts +653 -0
- package/src/engine/logger.ts +96 -0
- package/src/engine/loop.ts +5517 -0
- package/src/engine/parser.ts +347 -0
- package/src/engine/prompt.ts +860 -0
- package/src/engine/reports.ts +47 -0
- package/src/engine/stacks.ts +448 -0
- package/src/engine/types.ts +291 -0
- package/src/engine/usage.ts +186 -0
- package/src/github/app-auth.ts +161 -0
- package/src/github/cli.ts +448 -0
- package/src/github/installations.ts +133 -0
- package/src/github/pr.ts +321 -0
- package/src/github/provision.ts +118 -0
- package/src/github/push.ts +122 -0
- package/src/index-util.ts +50 -0
- package/src/index.ts +81 -0
- package/src/init/cli.ts +248 -0
- package/src/init/gitignore.ts +74 -0
- package/src/llm/ollama.ts +308 -0
- package/src/llm/openrouter.ts +868 -0
- package/src/llm/preflight.ts +367 -0
- package/src/llm/transcript-capture.ts +84 -0
- package/src/memory/embed.ts +110 -0
- package/src/memory/index.ts +22 -0
- package/src/memory/local.ts +259 -0
- package/src/memory/summarizer.ts +283 -0
- package/src/memory/types.ts +153 -0
- package/src/memory/uwuchat.ts +157 -0
- package/src/migrate/cli.ts +115 -0
- package/src/orchestrator/analyze-cli.ts +104 -0
- package/src/orchestrator/auto-split.ts +206 -0
- package/src/orchestrator/cleanup.ts +1003 -0
- package/src/orchestrator/cli.ts +3571 -0
- package/src/orchestrator/cost-estimate.ts +564 -0
- package/src/orchestrator/cost-history-cli.ts +242 -0
- package/src/orchestrator/cost-history.ts +397 -0
- package/src/orchestrator/git-sync.ts +250 -0
- package/src/orchestrator/index.ts +153 -0
- package/src/orchestrator/log-analysis.ts +0 -0
- package/src/orchestrator/merge-check.ts +108 -0
- package/src/orchestrator/pipeline.ts +411 -0
- package/src/orchestrator/resume.ts +1940 -0
- package/src/orchestrator/reviewer.ts +503 -0
- package/src/orchestrator/split.ts +296 -0
- package/src/orchestrator/state.ts +542 -0
- package/src/orchestrator/status.ts +697 -0
- package/src/orchestrator/verification-gate.ts +134 -0
- package/src/orchestrator/watch.ts +898 -0
- package/src/permissions/commands.ts +1083 -0
- package/src/permissions/config.ts +241 -0
- package/src/permissions/index.ts +12 -0
- package/src/permissions/protected-files.ts +96 -0
- package/src/permissions/store-protection.ts +272 -0
- package/src/project-store.ts +648 -0
- package/src/projects/cli.ts +382 -0
- package/src/qa/qa.ts +487 -0
- package/src/tools/browser/handler.ts +346 -0
- package/src/tools/browser/service.ts +406 -0
- package/src/tools/browser/smoke.ts +78 -0
- package/src/tools/browser/tool.ts +99 -0
- package/src/tools/executor.ts +2575 -0
- package/src/tools/language-detect.ts +183 -0
- package/src/tools/output-summarizer.ts +369 -0
- package/src/tools/run-tests.ts +302 -0
- package/src/tools/set-indentation-tool.ts +49 -0
- package/src/tools/test-selection.ts +160 -0
- package/src/vendor/tests/smoke.ts +103 -0
- package/src/vendor/zoo-code/VENDOR-NOTES.md +213 -0
- package/src/vendor/zoo-code/shim/anthropic.ts +71 -0
- package/src/vendor/zoo-code/shim/openai.d.ts +60 -0
- package/src/vendor/zoo-code/shim/os-name.ts +18 -0
- package/src/vendor/zoo-code/shim/strip-bom.ts +14 -0
- package/src/vendor/zoo-code/shim/vscode.ts +76 -0
- package/src/vendor/zoo-code/src/core/config/CustomModesManager.ts +1015 -0
- package/src/vendor/zoo-code/src/core/diff/strategies/multi-search-replace.ts +670 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/capabilities.ts +46 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/custom-instructions.ts +559 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/index.ts +10 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/markdown-formatting.ts +7 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/modes.ts +35 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/objective.ts +13 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/rules.ts +95 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/skills.ts +105 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/system-info.ts +30 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/tool-use-guidelines.ts +9 -0
- package/src/vendor/zoo-code/src/core/prompts/sections/tool-use.ts +7 -0
- package/src/vendor/zoo-code/src/core/prompts/system.ts +176 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/access_mcp_resource.ts +41 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/apply_diff.ts +40 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/apply_patch.ts +61 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/ask_followup_question.ts +62 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/attempt_completion.ts +33 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/codebase_search.ts +43 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/converters.ts +109 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/edit.ts +48 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/edit_file.ts +72 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/execute_command.ts +54 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/generate_image.ts +51 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/index.ts +75 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/list_files.ts +41 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/mcp_server.ts +75 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/new_task.ts +39 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/read_command_output.ts +81 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/read_file.ts +169 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/run_slash_command.ts +31 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/search_files.ts +50 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/search_replace.ts +51 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/skill.ts +33 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/switch_mode.ts +31 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/update_todo_list.ts +54 -0
- package/src/vendor/zoo-code/src/core/prompts/tools/native-tools/write_to_file.ts +40 -0
- package/src/vendor/zoo-code/src/core/prompts/types.ts +12 -0
- package/src/vendor/zoo-code/src/i18n/index.ts +19 -0
- package/src/vendor/zoo-code/src/integrations/misc/extract-text.ts +81 -0
- package/src/vendor/zoo-code/src/services/checkpoints/RepoPerTaskCheckpointService.ts +15 -0
- package/src/vendor/zoo-code/src/services/checkpoints/ShadowCheckpointService.ts +553 -0
- package/src/vendor/zoo-code/src/services/checkpoints/excludes.ts +212 -0
- package/src/vendor/zoo-code/src/services/checkpoints/index.ts +3 -0
- package/src/vendor/zoo-code/src/services/checkpoints/types.ts +35 -0
- package/src/vendor/zoo-code/src/services/code-index/manager.ts +19 -0
- package/src/vendor/zoo-code/src/services/mcp/McpHub.ts +36 -0
- package/src/vendor/zoo-code/src/services/roo-config/index.ts +441 -0
- package/src/vendor/zoo-code/src/services/search/file-search.ts +143 -0
- package/src/vendor/zoo-code/src/services/skills/SkillsManager.ts +20 -0
- package/src/vendor/zoo-code/src/shared/globalFileNames.ts +9 -0
- package/src/vendor/zoo-code/src/shared/language.ts +43 -0
- package/src/vendor/zoo-code/src/shared/modes.ts +257 -0
- package/src/vendor/zoo-code/src/shared/tools.ts +385 -0
- package/src/vendor/zoo-code/src/utils/fs.ts +39 -0
- package/src/vendor/zoo-code/src/utils/globalContext.ts +22 -0
- package/src/vendor/zoo-code/src/utils/json-schema.ts +16 -0
- package/src/vendor/zoo-code/src/utils/logging.ts +21 -0
- package/src/vendor/zoo-code/src/utils/mcp-name.ts +190 -0
- package/src/vendor/zoo-code/src/utils/object.ts +18 -0
- package/src/vendor/zoo-code/src/utils/path.ts +94 -0
- package/src/vendor/zoo-code/src/utils/shell.ts +376 -0
- package/src/vendor/zoo-code/src/utils/text-normalization.ts +99 -0
- package/src/vendor/zoo-code/types/global-settings.ts +19 -0
- package/src/vendor/zoo-code/types/index.ts +22 -0
- package/src/vendor/zoo-code/types/message.ts +375 -0
- package/src/vendor/zoo-code/types/mode.ts +241 -0
- package/src/vendor/zoo-code/types/todo.ts +19 -0
- package/src/vendor/zoo-code/types/tool-params.ts +116 -0
- package/src/vendor/zoo-code/types/tool.ts +67 -0
- package/src/vendor/zoo-code/types/vscode.ts +84 -0
- package/src/vision/describe.ts +242 -0
- package/src/vision/tool.ts +91 -0
- package/src/watcher/cli.ts +369 -0
- package/src/watcher/github.ts +304 -0
- package/src/watcher/index.ts +59 -0
- package/src/watcher/state.ts +254 -0
- package/src/watcher/watch.ts +562 -0
- package/tsconfig.json +18 -0
|
@@ -0,0 +1,512 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Token-budget-aware context condensation — Phase 3 of the history-
|
|
3
|
+
* management plan (replaces the sliding-window placeholder's information
|
|
4
|
+
* loss; see `truncateHistory`'s doc comment in src/engine/loop.ts, which
|
|
5
|
+
* explicitly called this out as the future token-based pass).
|
|
6
|
+
*
|
|
7
|
+
* ─── What this does ──────────────────────────────────────────────────────────
|
|
8
|
+
*
|
|
9
|
+
* When the LAST request's real prompt-token count (from
|
|
10
|
+
* `LlmResponse.usage.promptTokens` — the same number that drives
|
|
11
|
+
* `BudgetTracker`/`totalInputTokens`) crosses a configurable fraction of the
|
|
12
|
+
* model's real context window, the oldest complete turns are summarized by an
|
|
13
|
+
* LLM into ONE compact synthetic `role: "user"` message, and the recent,
|
|
14
|
+
* uncompressed tail is kept verbatim after it. This is strictly better than
|
|
15
|
+
* `truncateHistory`'s drop-oldest eviction for genuinely long sessions: the
|
|
16
|
+
* stale turns are compressed instead of destroyed, so the model does not
|
|
17
|
+
* silently lose (and later re-derive at full cost) earlier file reads,
|
|
18
|
+
* command outputs, or decisions.
|
|
19
|
+
*
|
|
20
|
+
* ─── Why `role: "user"` for the synthetic message ───────────────────────────
|
|
21
|
+
*
|
|
22
|
+
* The condensed message replaces a chunk of assistant/tool/user turns, so it
|
|
23
|
+
* must be a role the API accepts after a `tool` message and before another
|
|
24
|
+
* assistant turn. `role: "user"` is the only one of the four roles that is
|
|
25
|
+
* always legal there: `system` is only legal as message[0], and `assistant`
|
|
26
|
+
* / `tool` would break the tool-call-group protocol (DeepSeek's official
|
|
27
|
+
* endpoint 400s on orphaned `tool` messages — see truncateHistory's 2026-08-01
|
|
28
|
+
* fix). A `user` message is exactly how other harnesses (including Zoo Code's
|
|
29
|
+
* own rollback/summary paths) inject synthetic context, and it lets the
|
|
30
|
+
* model's next turn respond to the summary as new instructions.
|
|
31
|
+
*
|
|
32
|
+
* ─── Stable prefix / prompt caching (constraint 5) ──────────────────────────
|
|
33
|
+
*
|
|
34
|
+
* The caller (HeadlessSession) REPLACES its working message state with the
|
|
35
|
+
* condensed array, and tracks a `condensedUpTo` marker (2 = nothing condensed
|
|
36
|
+
* yet; 3 = the summary now sits at index 2). Re-condensation is then gated on
|
|
37
|
+
* the UNCONDENSED TAIL having grown past `MIN_CONDENSE_TAIL_GROWTH` messages
|
|
38
|
+
* since the last pass — so the sent prefix ([system, firstUser, summary, ...
|
|
39
|
+
* recent tail]) stays byte-identical across many subsequent calls and only
|
|
40
|
+
* changes at a re-condensation point, exactly like `truncateHistory`'s fixed
|
|
41
|
+
* batch eviction (batch, don't reslice every call).
|
|
42
|
+
*
|
|
43
|
+
* ─── Failure contract (non-fatal, matches the repo idiom) ───────────────────
|
|
44
|
+
*
|
|
45
|
+
* The condensation LLM call is an auxiliary subsystem: if it fails (timeout,
|
|
46
|
+
* provider error, unusable summary), `maybeCondense` resolves to null and the
|
|
47
|
+
* caller falls back to `truncateHistory` for that call. It never fails or
|
|
48
|
+
* blocks the session. The ONE thing that is NOT non-fatal is cost accounting:
|
|
49
|
+
* a successful condensation call's usage is fed into the same BudgetTracker +
|
|
50
|
+
* running totals as the main session (constraint 2 of
|
|
51
|
+
* plans/context-condensation.md), and a BudgetExceededError from that
|
|
52
|
+
* accounting propagates so the session aborts exactly as if a main call had
|
|
53
|
+
* tripped the cap.
|
|
54
|
+
*/
|
|
55
|
+
|
|
56
|
+
import type { ChatMessage, LlmClient, LlmRequest, LlmResponse } from "./types.js"
|
|
57
|
+
import type { Logger } from "./logger.js"
|
|
58
|
+
|
|
59
|
+
/**
|
|
60
|
+
* Default fraction of the model's context window at which condensation
|
|
61
|
+
* triggers. 0.75 (75%) balances two failure modes:
|
|
62
|
+
* - too LOW (e.g. 0.5): condensation fires on sessions that could have
|
|
63
|
+
* finished comfortably inside the window, paying an LLM call + latency for
|
|
64
|
+
* nothing (the whole point of keeping the sliding-window fallback);
|
|
65
|
+
* - too HIGH (e.g. 0.95): the model's real usable context is smaller than
|
|
66
|
+
* the advertised number (system prompt + tool schemas + the summary call's
|
|
67
|
+
* own request + a long reasoning generation all compete for the same
|
|
68
|
+
* window), so waiting that long risks a provider-side context-overflow
|
|
69
|
+
* error on the very next request.
|
|
70
|
+
* 0.75 leaves a full quarter of the window as headroom for the next
|
|
71
|
+
* iteration's growth after condensation, which (combined with the
|
|
72
|
+
* batch-once-stable semantics above) keeps the condensed prefix stable across
|
|
73
|
+
* many subsequent calls instead of re-condensing every turn.
|
|
74
|
+
*/
|
|
75
|
+
export const DEFAULT_CONDENSE_THRESHOLD_FRACTION = 0.75
|
|
76
|
+
|
|
77
|
+
/**
|
|
78
|
+
* Default fraction of the context window at which the ASYNC early-fire
|
|
79
|
+
* background condensation kicks off (plans/smart-condensation-async.md part
|
|
80
|
+
* 2). Must be BELOW the hard threshold fraction; at 0.6 vs a 0.75 hard
|
|
81
|
+
* threshold it leaves a 15-percentage-point runway for the 55-80s (measured)
|
|
82
|
+
* condensation call to resolve in the background while the main loop keeps
|
|
83
|
+
* making forward progress.
|
|
84
|
+
*
|
|
85
|
+
* PROVISIONAL: part 1's real per-iteration conversation-size data (the
|
|
86
|
+
* lastPromptTokens field added to the `[usage] running total` line) is what
|
|
87
|
+
* this number is supposed to be derived from — the conversation-size point
|
|
88
|
+
* where prompt-cache degradation becomes significant. That data requires a
|
|
89
|
+
* real long session, which is a later round's natural by-product; until it
|
|
90
|
+
* lands, 0.6 is a documented, TUNABLE interim value (--condense-early-fire),
|
|
91
|
+
* not a magic constant imported from anywhere. Re-derive it from fresh data
|
|
92
|
+
* before trusting it.
|
|
93
|
+
*/
|
|
94
|
+
export const DEFAULT_CONDENSE_EARLY_FIRE_FRACTION = 0.6
|
|
95
|
+
|
|
96
|
+
/**
|
|
97
|
+
* Conservative default context window (tokens) when the real value cannot be
|
|
98
|
+
* resolved from OpenRouter's models endpoint AND no override is configured.
|
|
99
|
+
* Deliberately conservative (deepseek/deepseek-v4-flash's official endpoint
|
|
100
|
+
* is 1,048,576 per the pricing pin — see src/budget/cost.ts), because the
|
|
101
|
+
* real window is only ever used to decide WHEN to condense; a conservative
|
|
102
|
+
* number triggers condensation earlier than strictly necessary, which is the
|
|
103
|
+
* safe direction. The live value is fetched via OpenRouter's
|
|
104
|
+
* `/api/v1/models/<id>/endpoints` the same way the pricing table was
|
|
105
|
+
* verified (see `fetchModelContextWindow` in src/llm/openrouter.ts).
|
|
106
|
+
*/
|
|
107
|
+
export const DEFAULT_CONTEXT_WINDOW_TOKENS = 128_000
|
|
108
|
+
|
|
109
|
+
/**
|
|
110
|
+
* Soft target cap on the condensed summary message, in characters. The model
|
|
111
|
+
* is asked to stay under this; the result is hard-truncated at
|
|
112
|
+
* `MAX_CONDENSED_MESSAGE_CHARS` regardless (a runaway summary must never
|
|
113
|
+
* blow the context budget this feature exists to protect).
|
|
114
|
+
*/
|
|
115
|
+
export const CONDENSE_TARGET_CHARS = 6_000
|
|
116
|
+
|
|
117
|
+
/** Hard cap on what the condensation response is ALLOWED to become. */
|
|
118
|
+
export const MAX_CONDENSED_MESSAGE_CHARS = 16_000
|
|
119
|
+
|
|
120
|
+
/**
|
|
121
|
+
* Default cap on the condensation call's OUTPUT tokens. Always sent on the
|
|
122
|
+
* wire (`max_tokens`) so a runaway reasoning generation can't burn unbounded
|
|
123
|
+
* tokens and then get rejected anyway by the `MAX_CONDENSED_MESSAGE_CHARS`
|
|
124
|
+
* post-generation check. 4096 is well above the 6,000-char target summary
|
|
125
|
+
* (`CONDENSE_TARGET_CHARS`) for typical output while bounding the worst case.
|
|
126
|
+
*/
|
|
127
|
+
export const DEFAULT_CONDENSE_MAX_TOKENS = 4096
|
|
128
|
+
|
|
129
|
+
/** Cap on the raw oldest-turn chunk fed to the condensation call. */
|
|
130
|
+
export const MAX_CONDENSE_INPUT_CHARS = 200_000
|
|
131
|
+
|
|
132
|
+
/**
|
|
133
|
+
* Minimum number of recent messages a condensation pass must LEAVE
|
|
134
|
+
* uncompressed. The model needs fresh context to continue; a summary alone is
|
|
135
|
+
* not enough. Also bounds how much a single pass can condense, which keeps
|
|
136
|
+
* re-condensation infrequent (the tail must re-grow past this before the next
|
|
137
|
+
* pass).
|
|
138
|
+
*/
|
|
139
|
+
export const MIN_KEPT_TAIL_MESSAGES = 10
|
|
140
|
+
|
|
141
|
+
/**
|
|
142
|
+
* A re-condensation is only allowed once the uncompressed tail has grown by
|
|
143
|
+
* this many messages past the previous condensation point (in addition to
|
|
144
|
+
* re-crossing the token threshold). This is the "batch, don't reslice"
|
|
145
|
+
* stability guarantee: between re-condensation points the sent prefix is
|
|
146
|
+
* byte-identical across calls, so provider-side prompt caching can accrue.
|
|
147
|
+
*/
|
|
148
|
+
export const MIN_CONDENSE_TAIL_GROWTH = 2 * MIN_KEPT_TAIL_MESSAGES
|
|
149
|
+
|
|
150
|
+
/** System prompt for the condensation LLM call. */
|
|
151
|
+
export const CONDENSE_SYSTEM_PROMPT =
|
|
152
|
+
"You are a conversation-compression engine for a software engineering agent. You will be given the " +
|
|
153
|
+
"OLDEST part of an agent session transcript (the model's earlier tool calls + their results + its " +
|
|
154
|
+
"notes). Your job is to compress it into a compact summary the agent can read INSTEAD of the original " +
|
|
155
|
+
"turns. HARD RULES:\n" +
|
|
156
|
+
"1. Preserve what a coding agent needs to continue WITHOUT re-deriving: files read and what was in them " +
|
|
157
|
+
"(paths + key contents verbatim where short), commands run and their outputs/errors (error messages " +
|
|
158
|
+
"VERBATIM), decisions made and why, facts learned, and anything marked IMPORTANT. Losing this forces " +
|
|
159
|
+
"the agent to re-read/re-run at real cost.\n" +
|
|
160
|
+
"2. NEVER invent or add content. If something is a guess, say it is a guess. Do not add conclusions the " +
|
|
161
|
+
"transcript does not support.\n" +
|
|
162
|
+
"3. Omit pure mechanics (e.g. a successful trivial command whose exact output no longer matters) — " +
|
|
163
|
+
"prefer a one-line note over a long quote.\n" +
|
|
164
|
+
"4. Output ONLY the compressed summary. No preamble like 'Here is the summary', no meta-commentary, " +
|
|
165
|
+
"no advice."
|
|
166
|
+
|
|
167
|
+
/** The synthetic message that stands in for the condensed chunk. */
|
|
168
|
+
export function buildCondensedMessage(summary: string): ChatMessage {
|
|
169
|
+
return {
|
|
170
|
+
role: "user",
|
|
171
|
+
content: `[Condensed summary of the earlier part of this session — read this INSTEAD of the turns it replaces; it is a compression, not a quote.]\n\n${summary}`,
|
|
172
|
+
}
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
/**
|
|
176
|
+
* Build the user prompt for the condensation call: the oldest chunk, clearly
|
|
177
|
+
* delimited, with an explicit target size.
|
|
178
|
+
*/
|
|
179
|
+
export function buildCondenseUserPrompt(chunk: ChatMessage[], targetChars: number): string {
|
|
180
|
+
// Deliberately omit `reasoning` from the transcript: it is disposable (the
|
|
181
|
+
// model already spent its generation on it) and reading it only inflates
|
|
182
|
+
// this call's own input cost. estimateMessageChars still counts it for
|
|
183
|
+
// SIZING — the summarizer just doesn't need to read it.
|
|
184
|
+
const transcript = chunk
|
|
185
|
+
.map((m) => {
|
|
186
|
+
const head =
|
|
187
|
+
m.role === "tool"
|
|
188
|
+
? `[tool result for ${m.name ?? "tool"}]`
|
|
189
|
+
: m.role === "assistant" && m.tool_calls
|
|
190
|
+
? `[assistant tool_calls: ${m.tool_calls.map((c) => c.function?.name ?? "?").join(", ")}]`
|
|
191
|
+
: `[${m.role}]`
|
|
192
|
+
const body =
|
|
193
|
+
typeof m.content === "string" && m.content.trim() !== ""
|
|
194
|
+
? m.content
|
|
195
|
+
: m.tool_calls
|
|
196
|
+
? m.tool_calls
|
|
197
|
+
.map((c) => `${c.function?.name ?? "?"}: ${(c.function?.arguments ?? "").slice(0, 400)}`)
|
|
198
|
+
.join("\n")
|
|
199
|
+
: "(no text)"
|
|
200
|
+
return `${head}\n${body}`
|
|
201
|
+
})
|
|
202
|
+
.join("\n\n")
|
|
203
|
+
|
|
204
|
+
return (
|
|
205
|
+
`Compress the OLDEST part of this agent session transcript (${chunk.length} messages). ` +
|
|
206
|
+
`Produce a summary of roughly ${targetChars} characters or fewer — enough that the agent can continue ` +
|
|
207
|
+
`without re-reading or re-running what is summarized. Keep error messages and important file/command ` +
|
|
208
|
+
`details verbatim.\n\n` +
|
|
209
|
+
`=== TRANSCRIPT BEGIN ===\n${transcript}\n=== TRANSCRIPT END ===`
|
|
210
|
+
)
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
/**
|
|
214
|
+
* Choose how many messages to condense from the front of `messages` (from
|
|
215
|
+
* `startIndex` onward — the caller passes 2 for a fresh condensation, which
|
|
216
|
+
* folds the existing summary at index 2 back in on re-condensation).
|
|
217
|
+
*
|
|
218
|
+
* `wantTokens` is the number of tokens the condensed chunk is allowed to have
|
|
219
|
+
* occupied before compression (derived from the context budget);
|
|
220
|
+
* `conservativeCharsPerToken` is a deliberate over-estimate (a condensation
|
|
221
|
+
* that frees TOO LITTLE is harmless — the next iteration simply re-checks;
|
|
222
|
+
* one that frees TOO MUCH would drop the tool-group boundary invariants).
|
|
223
|
+
*
|
|
224
|
+
* The boundary lands on a clean turn boundary: the returned count never
|
|
225
|
+
* splits an assistant `tool_calls` message from its `tool` response messages
|
|
226
|
+
* (the 2026-08-01 DeepSeek HTTP 400 class of bug — see truncateHistory's doc
|
|
227
|
+
* comment). Sharing this helper with `truncateHistory` (via
|
|
228
|
+
* `computeEvictCount`/`skipOrphanedToolMessages`) means both the message-count
|
|
229
|
+
* fallback and the token-aware path enforce the SAME invariant from ONE
|
|
230
|
+
* implementation family.
|
|
231
|
+
*/
|
|
232
|
+
export function computeCondenseCount(
|
|
233
|
+
messages: ChatMessage[],
|
|
234
|
+
wantTokens: number,
|
|
235
|
+
conservativeCharsPerToken = 8,
|
|
236
|
+
startIndex = 2,
|
|
237
|
+
): number {
|
|
238
|
+
if (messages.length < 4) {
|
|
239
|
+
return 0
|
|
240
|
+
}
|
|
241
|
+
const budgetChars = Math.max(1, wantTokens) * conservativeCharsPerToken
|
|
242
|
+
// Never condense the most recent `MIN_KEPT_TAIL_MESSAGES` messages (but
|
|
243
|
+
// allow a tiny history to be condensed down to zero tail — the summary is
|
|
244
|
+
// still strictly better than eviction).
|
|
245
|
+
const keepTail = Math.min(MIN_KEPT_TAIL_MESSAGES, Math.max(0, messages.length - 4))
|
|
246
|
+
const maxCount = Math.max(2, messages.length - startIndex - keepTail)
|
|
247
|
+
let count = 0
|
|
248
|
+
let chars = 0
|
|
249
|
+
for (let i = startIndex; i < messages.length && count < maxCount; i++) {
|
|
250
|
+
const m = messages[i]
|
|
251
|
+
const mChars = estimateMessageChars(m)
|
|
252
|
+
if (count > 0 && chars + mChars > budgetChars) {
|
|
253
|
+
break
|
|
254
|
+
}
|
|
255
|
+
count++
|
|
256
|
+
chars += mChars
|
|
257
|
+
if (m.role === "assistant" && m.tool_calls && m.tool_calls.length > 0) {
|
|
258
|
+
// Pull the whole tool group (the assistant call + its `tool`
|
|
259
|
+
// responses) into the chunk — never stop mid-group, even if that
|
|
260
|
+
// means the tail ends up a little smaller than keepTail.
|
|
261
|
+
let j = i + 1
|
|
262
|
+
while (j < messages.length && messages[j].role === "tool") {
|
|
263
|
+
count++
|
|
264
|
+
chars += estimateMessageChars(messages[j])
|
|
265
|
+
j++
|
|
266
|
+
}
|
|
267
|
+
i = j - 1
|
|
268
|
+
}
|
|
269
|
+
}
|
|
270
|
+
return count
|
|
271
|
+
}
|
|
272
|
+
|
|
273
|
+
/**
|
|
274
|
+
* Shared token-budget math for BOTH condensation paths: how many tokens the
|
|
275
|
+
* next request should target after condensation (half the hard threshold —
|
|
276
|
+
* enough that we don't re-cross it every single call) and how many oldest
|
|
277
|
+
* messages that corresponds to condensing (tool-call-group-safe, via
|
|
278
|
+
* `computeCondenseCount`). The synchronous path (`maybeCondense`) and the
|
|
279
|
+
* async early-fire path (HeadlessSession.fireBackgroundCondense) derive the
|
|
280
|
+
* SAME boundary from the SAME helper — never two independent implementations
|
|
281
|
+
* of the boundary rule.
|
|
282
|
+
*/
|
|
283
|
+
export function computeCondensePlan(
|
|
284
|
+
messages: ChatMessage[],
|
|
285
|
+
lastPromptTokens: number,
|
|
286
|
+
contextWindowTokens: number,
|
|
287
|
+
thresholdFraction: number,
|
|
288
|
+
): { targetTokens: number; wantTokens: number; count: number } {
|
|
289
|
+
const targetTokens = Math.max(1, Math.floor(contextWindowTokens * thresholdFraction * 0.5))
|
|
290
|
+
const wantTokens = Math.max(1, lastPromptTokens - targetTokens)
|
|
291
|
+
const count = computeCondenseCount(messages, wantTokens)
|
|
292
|
+
return { targetTokens, wantTokens, count }
|
|
293
|
+
}
|
|
294
|
+
|
|
295
|
+
/**
|
|
296
|
+
* Shared tool-call-group-safe count computation for `truncateHistory`'s
|
|
297
|
+
* message-count eviction: how many messages to drop from index 2 onward
|
|
298
|
+
* (returned as a COUNT to drop, so the caller keeps system+firstUser).
|
|
299
|
+
*
|
|
300
|
+
* `evictCount` here means "drop this many of `rest`" — i.e. the kept tail
|
|
301
|
+
* starts at `rest.slice(evictCount)` — matching the existing
|
|
302
|
+
* `truncateHistory` contract exactly. The batch rounding (ceil to the next
|
|
303
|
+
* multiple of `batchSize`) is what keeps the sent prefix stable between
|
|
304
|
+
* eviction points.
|
|
305
|
+
*/
|
|
306
|
+
export function computeEvictCount(restLength: number, windowSize: number, batchSize: number): number {
|
|
307
|
+
const overflow = restLength - (windowSize - 2)
|
|
308
|
+
let evictCount = Math.ceil(overflow / batchSize) * batchSize
|
|
309
|
+
if (evictCount <= 0) {
|
|
310
|
+
evictCount = 0
|
|
311
|
+
}
|
|
312
|
+
return evictCount
|
|
313
|
+
}
|
|
314
|
+
|
|
315
|
+
/** Skip forward over leading `tool` messages (post-cut tail-start safety). */
|
|
316
|
+
export function skipOrphanedToolMessages(messages: ChatMessage[], from: number): number {
|
|
317
|
+
let idx = from
|
|
318
|
+
while (idx < messages.length && messages[idx].role === "tool") {
|
|
319
|
+
idx++
|
|
320
|
+
}
|
|
321
|
+
return idx
|
|
322
|
+
}
|
|
323
|
+
|
|
324
|
+
/** Rough character-length estimate of one message (cheap, monotonic). */
|
|
325
|
+
export function estimateMessageChars(m: ChatMessage): number {
|
|
326
|
+
let chars = typeof m.content === "string" ? m.content.length : 0
|
|
327
|
+
// Reasoning is echoed onto outgoing history and costs real prompt tokens —
|
|
328
|
+
// count it or sizing under-estimates reasoning-heavy sessions.
|
|
329
|
+
chars += m.reasoning?.length ?? 0
|
|
330
|
+
if (m.tool_calls) {
|
|
331
|
+
for (const c of m.tool_calls) {
|
|
332
|
+
chars += (c.function?.name?.length ?? 0) + (c.function?.arguments?.length ?? 0) + 8
|
|
333
|
+
}
|
|
334
|
+
}
|
|
335
|
+
return chars + 4
|
|
336
|
+
}
|
|
337
|
+
|
|
338
|
+
/**
|
|
339
|
+
* Run one condensation LLM call against `llmClient` (the session's client —
|
|
340
|
+
* same endpoint/auth, possibly a cheaper model id; see `condenseModel` in
|
|
341
|
+
* loop.ts). Returns the summary text. Throws on ANY failure — the caller
|
|
342
|
+
* (`maybeCondense`) catches and falls back to `truncateHistory`.
|
|
343
|
+
*
|
|
344
|
+
* The returned usage is surfaced via the `onUsage` callback so the session
|
|
345
|
+
* can feed it into the SAME BudgetTracker + running totals as the main
|
|
346
|
+
* session (constraint 2 — condensation must never be invisible spend).
|
|
347
|
+
*/
|
|
348
|
+
export async function condenseOldestTurns(
|
|
349
|
+
llmClient: LlmClient,
|
|
350
|
+
options: {
|
|
351
|
+
messages: ChatMessage[]
|
|
352
|
+
count: number
|
|
353
|
+
model: string
|
|
354
|
+
maxTokens?: number
|
|
355
|
+
signal?: AbortSignal
|
|
356
|
+
onUsage?: (usage: NonNullable<LlmResponse["usage"]>) => void
|
|
357
|
+
},
|
|
358
|
+
): Promise<string> {
|
|
359
|
+
const chunk = options.messages.slice(2, 2 + options.count)
|
|
360
|
+
const request: LlmRequest = {
|
|
361
|
+
model: options.model,
|
|
362
|
+
messages: [
|
|
363
|
+
{ role: "system", content: CONDENSE_SYSTEM_PROMPT },
|
|
364
|
+
{ role: "user", content: buildCondenseUserPrompt(chunk, CONDENSE_TARGET_CHARS) },
|
|
365
|
+
],
|
|
366
|
+
maxTokens: options.maxTokens,
|
|
367
|
+
signal: options.signal,
|
|
368
|
+
}
|
|
369
|
+
const response = await llmClient.createChatCompletion(request)
|
|
370
|
+
if (response.usage) {
|
|
371
|
+
options.onUsage?.(response.usage)
|
|
372
|
+
}
|
|
373
|
+
const text = response.message.content
|
|
374
|
+
if (typeof text !== "string" || text.trim() === "") {
|
|
375
|
+
throw new Error("condensation returned an empty message content")
|
|
376
|
+
}
|
|
377
|
+
const trimmed = text.trim()
|
|
378
|
+
if (trimmed.length > MAX_CONDENSED_MESSAGE_CHARS) {
|
|
379
|
+
throw new Error(
|
|
380
|
+
`condensation produced ${trimmed.length} chars (cap ${MAX_CONDENSED_MESSAGE_CHARS}) — falling back to message-count truncation`,
|
|
381
|
+
)
|
|
382
|
+
}
|
|
383
|
+
return trimmed
|
|
384
|
+
}
|
|
385
|
+
|
|
386
|
+
/**
|
|
387
|
+
* Non-fatal wrapper used by HeadlessSession: decide whether to condense and
|
|
388
|
+
* do it. Returns the REPLACEMENT messages array (the oldest chunk replaced by
|
|
389
|
+
* one synthetic summary at index 2) when condensation ran; null when it did
|
|
390
|
+
* not run (below threshold, prefix still stable, or the call failed and the
|
|
391
|
+
* fallback path should apply).
|
|
392
|
+
*
|
|
393
|
+
* The caller MUST replace its working state with the returned array and keep
|
|
394
|
+
* `condensedUpTo.value` (mutated here) so the next call sees the already-
|
|
395
|
+
* condensed prefix. `condenseInFlight.value` is mutated to true while the LLM
|
|
396
|
+
* call is running and reset to false when it completes/fails — it guarantees
|
|
397
|
+
* only one condensation call is ever in flight.
|
|
398
|
+
*/
|
|
399
|
+
export async function maybeCondense(options: {
|
|
400
|
+
llmClient: LlmClient
|
|
401
|
+
logger: Pick<Logger, "info" | "warn">
|
|
402
|
+
messages: ChatMessage[]
|
|
403
|
+
model: string
|
|
404
|
+
lastPromptTokens: number
|
|
405
|
+
contextWindowTokens: number
|
|
406
|
+
thresholdFraction: number
|
|
407
|
+
condensedUpTo: { value: number }
|
|
408
|
+
condenseInFlight: { value: boolean }
|
|
409
|
+
condenseModel: string
|
|
410
|
+
condenseMaxTokens?: number
|
|
411
|
+
condenseAbortSignal?: AbortSignal
|
|
412
|
+
onUsage?: (usage: NonNullable<LlmResponse["usage"]>) => void
|
|
413
|
+
}): Promise<ChatMessage[] | null> {
|
|
414
|
+
const {
|
|
415
|
+
llmClient,
|
|
416
|
+
logger,
|
|
417
|
+
messages,
|
|
418
|
+
lastPromptTokens,
|
|
419
|
+
contextWindowTokens,
|
|
420
|
+
thresholdFraction,
|
|
421
|
+
condensedUpTo,
|
|
422
|
+
condenseInFlight,
|
|
423
|
+
condenseModel,
|
|
424
|
+
condenseMaxTokens,
|
|
425
|
+
condenseAbortSignal,
|
|
426
|
+
onUsage,
|
|
427
|
+
} = options
|
|
428
|
+
|
|
429
|
+
// A failed/stale last request has no usable token count — never condense
|
|
430
|
+
// on guesswork.
|
|
431
|
+
if (!(lastPromptTokens > 0)) {
|
|
432
|
+
return null
|
|
433
|
+
}
|
|
434
|
+
const threshold = Math.max(1, contextWindowTokens * thresholdFraction)
|
|
435
|
+
if (lastPromptTokens < threshold) {
|
|
436
|
+
return null
|
|
437
|
+
}
|
|
438
|
+
// Prefix stability: after a condensation, the summary sits at index 2 and
|
|
439
|
+
// the tail must GROW meaningfully past it before we re-condense (otherwise
|
|
440
|
+
// the sent prefix would be rewritten on every call, defeating provider-
|
|
441
|
+
// side prompt caching — constraint 5). The FIRST condensation is exempt
|
|
442
|
+
// (there is no prefix to preserve yet).
|
|
443
|
+
const alreadyCondensed = condensedUpTo.value > 2
|
|
444
|
+
if (alreadyCondensed && messages.length - condensedUpTo.value < MIN_CONDENSE_TAIL_GROWTH) {
|
|
445
|
+
return null
|
|
446
|
+
}
|
|
447
|
+
if (condenseInFlight.value) {
|
|
448
|
+
return null
|
|
449
|
+
}
|
|
450
|
+
condenseInFlight.value = true
|
|
451
|
+
try {
|
|
452
|
+
// Condense enough that the next request lands comfortably below the
|
|
453
|
+
// threshold (target = half the threshold), so we don't re-cross it
|
|
454
|
+
// every single call. MIN_KEPT_TAIL_MESSAGES in computeCondenseCount
|
|
455
|
+
// keeps the recent tail verbatim regardless. Same shared math as the
|
|
456
|
+
// async early-fire path (computeCondensePlan).
|
|
457
|
+
const { count } = computeCondensePlan(messages, lastPromptTokens, contextWindowTokens, thresholdFraction)
|
|
458
|
+
if (count < 2) {
|
|
459
|
+
return null
|
|
460
|
+
}
|
|
461
|
+
logger.info("[condense] crossing token threshold — summarizing oldest turns", {
|
|
462
|
+
lastPromptTokens,
|
|
463
|
+
contextWindowTokens,
|
|
464
|
+
threshold: Math.round(threshold),
|
|
465
|
+
count,
|
|
466
|
+
condenseModel,
|
|
467
|
+
})
|
|
468
|
+
let summary: string
|
|
469
|
+
try {
|
|
470
|
+
summary = await condenseOldestTurns(llmClient, {
|
|
471
|
+
messages,
|
|
472
|
+
count,
|
|
473
|
+
// condenseModel (NOT model): the whole point of the `_condensation`
|
|
474
|
+
// mode-models key / --condense-model flag is that this call may use
|
|
475
|
+
// a cheaper model than the session's — and recordCondensationUsage
|
|
476
|
+
// already accounts its cost under condenseModel. Routing the call
|
|
477
|
+
// to `model` silently ignored the flag (call + accounting disagreed).
|
|
478
|
+
model: condenseModel,
|
|
479
|
+
maxTokens: condenseMaxTokens,
|
|
480
|
+
signal: condenseAbortSignal,
|
|
481
|
+
onUsage,
|
|
482
|
+
})
|
|
483
|
+
} catch (error) {
|
|
484
|
+
// Non-fatal (matches the repo's auxiliary-subsystem idiom): a
|
|
485
|
+
// failed condensation call falls back to truncateHistory for this
|
|
486
|
+
// call. The ONE exception is a BudgetExceededError thrown by
|
|
487
|
+
// onUsage's accounting (recordCondensationUsage in loop.ts) — that
|
|
488
|
+
// must propagate so the session aborts exactly as if a main call
|
|
489
|
+
// had tripped the cap, never silently under-report spend.
|
|
490
|
+
if (error instanceof Error && error.name === "BudgetExceededError") {
|
|
491
|
+
throw error
|
|
492
|
+
}
|
|
493
|
+
logger.warn("[condense] condensation call failed (non-fatal; falling back to message-count truncation)", {
|
|
494
|
+
error: error instanceof Error ? error.message : String(error),
|
|
495
|
+
})
|
|
496
|
+
return null
|
|
497
|
+
}
|
|
498
|
+
// The summary always lands at index 2 (right after system+firstUser);
|
|
499
|
+
// on re-condensation it FOLDS the previous summary back in, so there
|
|
500
|
+
// is never more than one summary message in the array.
|
|
501
|
+
condensedUpTo.value = 3
|
|
502
|
+
const condensed: ChatMessage[] = [messages[0], messages[1], buildCondensedMessage(summary), ...messages.slice(2 + count)]
|
|
503
|
+
logger.info("[condense] oldest turns condensed into one summary message", {
|
|
504
|
+
condensedCount: count,
|
|
505
|
+
historyBefore: messages.length,
|
|
506
|
+
historyAfter: condensed.length,
|
|
507
|
+
})
|
|
508
|
+
return condensed
|
|
509
|
+
} finally {
|
|
510
|
+
condenseInFlight.value = false
|
|
511
|
+
}
|
|
512
|
+
}
|