@ponythewhite/base-context 1.0.11 → 1.0.13
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +15 -0
- package/dist/base-context-runtime/pyproject.toml +1 -1
- package/dist/base-context-runtime/uv.lock +1 -1
- package/dist/build-info.json +1 -1
- package/dist/bundle/amazon-bedrock.js +3 -3
- package/dist/bundle/{anthropic-5PSCZEER.js → anthropic-BJVPUMPB.js} +2 -2
- package/dist/bundle/{azure-openai-responses-AYTLJR33.js → azure-openai-responses-ERRMAXK6.js} +4 -5
- package/dist/bundle/{bundled-modules-PEFT72QN.js → bundled-modules-URKVA3UB.js} +5 -5
- package/dist/bundle/{chunk-D3KOF57W.js → chunk-25VYRAC6.js} +14 -1
- package/dist/bundle/{chunk-2IPDSTDR.js → chunk-AEMRBMI2.js} +9 -9
- package/dist/bundle/{chunk-Q3MBSBEB.js → chunk-FOGKLC7V.js} +1 -1
- package/dist/bundle/{chunk-KT2BNEPE.js → chunk-MUFZODP5.js} +1 -1
- package/dist/bundle/{chunk-QKIFBA7N.js → chunk-Q6UAHLII.js} +2 -2
- package/dist/bundle/{chunk-TOOVREMP.js → chunk-TKOBEJ5E.js} +515 -107
- package/dist/bundle/{chunk-JZHGOOFM.js → chunk-TXVF6Q5H.js} +118 -22
- package/dist/bundle/{cli-main-BRXGYTGK.js → cli-main-ZTJM5EEH.js} +4 -4
- package/dist/bundle/cli.js +1 -1
- package/dist/bundle/{google-HTMCUJ7X.js → google-4GA4YMUX.js} +2 -2
- package/dist/bundle/{google-vertex-ZQO65AXF.js → google-vertex-7AMALD7D.js} +2 -2
- package/dist/bundle/{main-POUXGEL6.js → main-53I74INO.js} +5 -5
- package/dist/bundle/{mistral-SB7LU56G.js → mistral-74BXMNRM.js} +1 -1
- package/dist/bundle/{openai-codex-responses-5CEW4VN3.js → openai-codex-responses-COXNC4IH.js} +2 -2
- package/dist/bundle/{openai-completions-BAZJQSS2.js → openai-completions-BCSA5Q6S.js} +2 -2
- package/dist/bundle/{openai-responses-VHOE427P.js → openai-responses-ONCIZSJS.js} +3 -3
- package/dist/core/agent-session.d.ts +6 -0
- package/dist/core/agent-session.d.ts.map +1 -1
- package/dist/core/agent-session.js +362 -103
- package/dist/core/agent-session.js.map +1 -1
- package/dist/core/compaction/compaction.d.ts +9 -1
- package/dist/core/compaction/compaction.d.ts.map +1 -1
- package/dist/core/compaction/compaction.js +29 -3
- package/dist/core/compaction/compaction.js.map +1 -1
- package/dist/core/context-tree.d.ts +13 -10
- package/dist/core/context-tree.d.ts.map +1 -1
- package/dist/core/context-tree.js +49 -3
- package/dist/core/context-tree.js.map +1 -1
- package/dist/core/inference-coordinator.d.ts +1 -1
- package/dist/core/inference-coordinator.d.ts.map +1 -1
- package/dist/core/inference-coordinator.js +18 -5
- package/dist/core/inference-coordinator.js.map +1 -1
- package/dist/core/messages.d.ts +2 -0
- package/dist/core/messages.d.ts.map +1 -1
- package/dist/core/messages.js +2 -0
- package/dist/core/messages.js.map +1 -1
- package/dist/core/refinement/refinement.d.ts +2 -0
- package/dist/core/refinement/refinement.d.ts.map +1 -1
- package/dist/core/refinement/refinement.js +10 -1
- package/dist/core/refinement/refinement.js.map +1 -1
- package/dist/core/request-usage.d.ts +34 -0
- package/dist/core/request-usage.d.ts.map +1 -0
- package/dist/core/request-usage.js +111 -0
- package/dist/core/request-usage.js.map +1 -0
- package/dist/core/settings-manager.d.ts +3 -0
- package/dist/core/settings-manager.d.ts.map +1 -1
- package/dist/core/settings-manager.js +7 -0
- package/dist/core/settings-manager.js.map +1 -1
- package/dist/core/system-prompt.d.ts +2 -0
- package/dist/core/system-prompt.d.ts.map +1 -1
- package/dist/core/system-prompt.js +4 -2
- package/dist/core/system-prompt.js.map +1 -1
- package/dist/modes/agent-connection/daemon-agent-connection.d.ts.map +1 -1
- package/dist/modes/agent-connection/daemon-agent-connection.js +8 -1
- package/dist/modes/agent-connection/daemon-agent-connection.js.map +1 -1
- package/dist/modes/daemon/daemon-protocol.d.ts +13 -3
- package/dist/modes/daemon/daemon-protocol.d.ts.map +1 -1
- package/dist/modes/daemon/daemon-protocol.js +10 -2
- package/dist/modes/daemon/daemon-protocol.js.map +1 -1
- package/dist/modes/daemon/daemon-session-summarizer.d.ts +3 -0
- package/dist/modes/daemon/daemon-session-summarizer.d.ts.map +1 -1
- package/dist/modes/daemon/daemon-session-summarizer.js +71 -32
- package/dist/modes/daemon/daemon-session-summarizer.js.map +1 -1
- package/dist/modes/interactive/components/context-tree-format.d.ts.map +1 -1
- package/dist/modes/interactive/components/context-tree-format.js +49 -0
- package/dist/modes/interactive/components/context-tree-format.js.map +1 -1
- package/docs/compaction.md +57 -4
- package/docs/context-management.md +23 -1
- package/docs/rlm.md +1 -1
- package/docs/settings.md +21 -0
- package/docs/usage.md +24 -1
- package/package.json +4 -4
package/docs/compaction.md
CHANGED
|
@@ -35,16 +35,67 @@ or loss from a summary alone.
|
|
|
35
35
|
|
|
36
36
|
### When It Triggers
|
|
37
37
|
|
|
38
|
-
Auto-compaction
|
|
38
|
+
Auto-compaction uses 90% of the model context window as its default soft target:
|
|
39
39
|
|
|
40
40
|
```
|
|
41
|
-
|
|
41
|
+
softTarget = floor(0.9 * contextWindow)
|
|
42
|
+
threshold = min(contextWindow - reserveTokens, max(softTarget, fixedContextTokens + 4 * keepRecentTokens))
|
|
43
|
+
contextTokens > threshold
|
|
42
44
|
```
|
|
43
45
|
|
|
44
|
-
|
|
46
|
+
With default `keepRecentTokens` of 20000 and `reserveTokens` of 16384, before
|
|
47
|
+
fixed-context headroom raises the target, the thresholds are 244800 for a
|
|
48
|
+
272000-token model, 111616 for a 128000-token model, and 47616 for a 64000-token
|
|
49
|
+
model. The model-window reserve can therefore trigger compaction before 90%.
|
|
50
|
+
The model-window ceiling always applies.
|
|
51
|
+
|
|
52
|
+
`fixedContextTokens` estimates current system instructions, tool schemas, the
|
|
53
|
+
current TaskFrame and latest harness snapshot. It does not count all historical
|
|
54
|
+
messages or obsolete snapshots as fixed. Reserving room above this context avoids
|
|
55
|
+
repeated ineffective summaries when required instructions already exceed the
|
|
56
|
+
soft target. These are local estimates, not exact provider token counts.
|
|
57
|
+
|
|
58
|
+
Set `compaction.targetTokens` to a positive safe integer to replace the soft
|
|
59
|
+
target; required-context headroom can still raise a numeric target. Set
|
|
60
|
+
`"model-limit"` to use exactly `contextWindow - reserveTokens` instead.
|
|
61
|
+
Configure these settings in `~/.base-context/settings.json` or
|
|
62
|
+
`<project-dir>/.base-context/settings.json`.
|
|
63
|
+
|
|
64
|
+
This is a compaction heuristic, not a strict request cap, a validated backend
|
|
65
|
+
limit, or a promise of optimal token use. Earlier summaries trade shorter replay
|
|
66
|
+
for summary calls and possible rereads. Required instructions and replay
|
|
67
|
+
dependencies are not clipped to meet this target. Explicit SDK request-token
|
|
68
|
+
budgets remain separate; see [context management](context-management.md#model-aware-budgets).
|
|
45
69
|
|
|
46
70
|
You can also trigger manually with `/compact [instructions]`, where optional instructions focus the summary — for example `/compact focus on the auth refactor, remember the exact migration command`. The instructions are passed to the summarization prompt with high priority, persisted on the `CompactionEntry`, and shown on the `[compaction]` message in the TUI.
|
|
47
71
|
|
|
72
|
+
### Agent-Requested Compaction
|
|
73
|
+
|
|
74
|
+
The agent can request compaction before the automatic threshold, including when
|
|
75
|
+
using Astra. The bundled `compact` skill is enabled by default and runs from the
|
|
76
|
+
Python REPL:
|
|
77
|
+
|
|
78
|
+
```python
|
|
79
|
+
await compact.status()
|
|
80
|
+
await compact.run("keep the remaining plan and exact test names")
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
`status()` reports `tokens`, `context_window`, `percent`, and `scheduled`. Usage
|
|
84
|
+
can be unknown immediately after compaction. `run()` schedules compaction at the
|
|
85
|
+
next turn boundary, not in the middle of the Python cell. It returns
|
|
86
|
+
`{"scheduled": True}` when accepted, or `{"scheduled": False, "reason": ...}`
|
|
87
|
+
when no turn is active or there is nothing to compact yet. Optional instructions
|
|
88
|
+
focus the summary. Repeating the request before the boundary updates them.
|
|
89
|
+
Interrupted tool work can continue after the checkpoint; this does not start new
|
|
90
|
+
work after an otherwise completed turn.
|
|
91
|
+
|
|
92
|
+
This request does not depend on the automatic threshold or `compaction.enabled`.
|
|
93
|
+
It requires context optimization to be on. `compaction.agentCallable: false`
|
|
94
|
+
disables the skill, and disabling Python or bundled skills can remove its normal
|
|
95
|
+
entry point. Each recursive agent uses its own session's compaction path; children
|
|
96
|
+
inherit the parent's compact-skill availability. There is no Astra-specific gate.
|
|
97
|
+
A request compacts the requesting session, not the whole agent tree.
|
|
98
|
+
|
|
48
99
|
### How It Works
|
|
49
100
|
|
|
50
101
|
1. **Find cut point**: Walk backwards from newest message, accumulating token estimates until `keepRecentTokens` (default 20k, configurable in `~/.base-context/settings.json` or `<project-dir>/.base-context/settings.json`) is reached
|
|
@@ -432,9 +483,11 @@ Configure compaction in `~/.base-context/settings.json` or `<project-dir>/.base-
|
|
|
432
483
|
| `enabled` | `true` | Enable auto-compaction |
|
|
433
484
|
| `reserveTokens` | `16384` | Headroom used by the compaction threshold |
|
|
434
485
|
| `keepRecentTokens` | `20000` | Estimated recent-token target for the retained tail |
|
|
486
|
+
| `targetTokens` | 90% of the model context window | Positive safe integer to override the soft target, or `"model-limit"` for the model-window threshold; fixed-context headroom and the model reserve still apply as described above |
|
|
487
|
+
| `agentCallable` | `true` | Expose the `compact` skill so the agent can request earlier compaction |
|
|
435
488
|
|
|
436
489
|
Disable automatic compaction with `"compaction": { "enabled": false }`. Manual
|
|
437
|
-
`/compact`
|
|
490
|
+
`/compact` and agent-requested compaction remain available while `context.mode` is `"on"`. Setting `context.mode`
|
|
438
491
|
to `"off"` disables context optimization, including manual compaction; logging,
|
|
439
492
|
recovery, limits and other non-optimization ownership remain active.
|
|
440
493
|
|
|
@@ -82,7 +82,29 @@ Profiles identify the API, provider, endpoint, final model, context allowance, o
|
|
|
82
82
|
|
|
83
83
|
In supported native Responses/Codex selection paths, historical assistant literals can become optional at accepted boundaries. Users, the latest assistant, TaskFrames, summaries, recovery, and required dependencies remain mandatory. If that set cannot fit, the request can refuse. Unsupported media, opaque layouts, or missing contracts also remain explicit limits.
|
|
84
84
|
|
|
85
|
-
See [SDK request-token profiles](sdk.md#explicit-request-token-budget-profiles) for the exact scope and configuration.
|
|
85
|
+
See [SDK request-token profiles](sdk.md#explicit-request-token-budget-profiles) for the exact scope and configuration.
|
|
86
|
+
|
|
87
|
+
Ordinary [compaction](compaction.md) uses a separate soft target of 90% of the
|
|
88
|
+
model context window: `floor(0.9 * contextWindow)`. The actual trigger also leaves
|
|
89
|
+
`4 * keepRecentTokens` above estimated fixed context, then caps the result at
|
|
90
|
+
`contextWindow - reserveTokens`. Fixed context includes current system
|
|
91
|
+
instructions, tool schemas, the current TaskFrame and latest harness snapshot,
|
|
92
|
+
not every historical message. This avoids repeated ineffective summaries of
|
|
93
|
+
noncompactable instructions.
|
|
94
|
+
|
|
95
|
+
With small fixed context, default thresholds are 244800 estimated tokens for a
|
|
96
|
+
272000-token model, 111616 for a 128000-token model, and 47616 for a 64000-token
|
|
97
|
+
model. The model-window reserve can trigger compaction before 90%. Set
|
|
98
|
+
`compaction.targetTokens` to a positive safe integer to replace the soft target
|
|
99
|
+
(still subject to fixed-context headroom), or `"model-limit"` to use exactly
|
|
100
|
+
`contextWindow - reserveTokens`. Required replay groups are not clipped to meet
|
|
101
|
+
this target. Summaries can add calls and rereads; this is not a strict request
|
|
102
|
+
cap, an optimality guarantee, or evidence of a deployment's actual limit.
|
|
103
|
+
|
|
104
|
+
The agent can request an earlier checkpoint from the Python REPL with
|
|
105
|
+
`await compact.run()`, independently of the automatic threshold. It can inspect
|
|
106
|
+
usage with `await compact.status()`. This is enabled by default for all models,
|
|
107
|
+
including Astra; see [agent-requested compaction](compaction.md#agent-requested-compaction).
|
|
86
108
|
|
|
87
109
|
## Stable context epochs
|
|
88
110
|
|
package/docs/rlm.md
CHANGED
|
@@ -101,7 +101,7 @@ await agent_message.send(
|
|
|
101
101
|
|
|
102
102
|
#### Child handles and lifecycle
|
|
103
103
|
|
|
104
|
-
An admission handle contains `rlm_child_id`, `name`, `session_dir`, and `model`. Child usage is attributed to the parent session while remaining distinguishable in context-tree reporting.
|
|
104
|
+
An admission handle contains `rlm_child_id`, `name`, `session_dir`, and `model`. Child usage is attributed to the parent session while remaining distinguishable in context-tree reporting. Native child usage reporting continues after passivation and rehydration. `/context` separately totals saved request receipts for the captured family; parent attribution records are not added to those totals. See [token usage and status](usage.md#token-usage-and-status) for missing-data, catalog-estimate, and goal-budget scope.
|
|
105
105
|
|
|
106
106
|
The parent-scoped child registry survives compaction, kernel restart, and parent restoration:
|
|
107
107
|
|
package/docs/settings.md
CHANGED
|
@@ -94,6 +94,8 @@ Private download manifests require `version` and `package` (or `packageName`) se
|
|
|
94
94
|
| `compaction.enabled` | boolean | `true` | Enable auto-compaction |
|
|
95
95
|
| `compaction.reserveTokens` | number | `16384` | Tokens reserved for LLM response |
|
|
96
96
|
| `compaction.keepRecentTokens` | number | `20000` | Recent tokens to keep (not summarized) |
|
|
97
|
+
| `compaction.targetTokens` | positive integer or `"model-limit"` | 90% of the model context window | Override the soft compaction target, still capped by the model window minus reserve |
|
|
98
|
+
| `compaction.agentCallable` | boolean | `true` | Expose the `compact` skill so the agent can request compaction before the automatic threshold |
|
|
97
99
|
| `compaction.model` | object | Current main model and effort | Explicit summary model: `provider`, `modelId`, and `thinkingLevel` are all required |
|
|
98
100
|
|
|
99
101
|
```json
|
|
@@ -107,6 +109,25 @@ Private download manifests require `version` and `package` (or `packageName`) se
|
|
|
107
109
|
```
|
|
108
110
|
|
|
109
111
|
|
|
112
|
+
Without `targetTokens`, the soft target is 90% of the model context window:
|
|
113
|
+
`floor(0.9 * contextWindow)`. The trigger also leaves `4 * keepRecentTokens` above
|
|
114
|
+
an estimate of current fixed instructions, tool schemas, TaskFrame and latest
|
|
115
|
+
harness snapshot. It is capped at `contextWindow - reserveTokens`. With small
|
|
116
|
+
fixed context, default thresholds are 244800 for a 272000-token model, 111616 for a
|
|
117
|
+
128000-token model, and 47616 for a 64000-token model. The model-window reserve can
|
|
118
|
+
therefore trigger compaction before 90%. A positive safe integer replaces the
|
|
119
|
+
soft target, but required-context headroom can raise it; `"model-limit"` uses
|
|
120
|
+
exactly `contextWindow - reserveTokens`. This is a summary heuristic, not strict
|
|
121
|
+
provider-request admission. See [compaction](compaction.md#when-it-triggers) and
|
|
122
|
+
[model-aware budgets](context-management.md#model-aware-budgets).
|
|
123
|
+
|
|
124
|
+
With the default `compaction.agentCallable: true`, the agent can use
|
|
125
|
+
`await compact.status()` and `await compact.run(instructions=None)` in the Python
|
|
126
|
+
REPL. This model-independent path includes Astra and can request compaction before
|
|
127
|
+
the automatic threshold. It also works with `compaction.enabled: false`, but new
|
|
128
|
+
requests require `context.mode: "on"`. Requests run at a turn boundary, not
|
|
129
|
+
mid-cell. See [agent-requested compaction](compaction.md#agent-requested-compaction).
|
|
130
|
+
|
|
110
131
|
Set `compaction.model` to choose the model and effort for manual, automatic and
|
|
111
132
|
model-requested compaction summaries. For example, when this exact model/route is
|
|
112
133
|
configured and budgeted:
|
package/docs/usage.md
CHANGED
|
@@ -49,7 +49,7 @@ Type `/` in the editor to open command completion. Extensions can register custo
|
|
|
49
49
|
| `/new` | Start a new session |
|
|
50
50
|
| `/name <name>` | Set session display name |
|
|
51
51
|
| `/session` | Show session file, ID, and message counts |
|
|
52
|
-
| `/usage`, `/context` | Show
|
|
52
|
+
| `/usage`, `/context` | Show captured-family request usage, context, and separate goal-budget scope |
|
|
53
53
|
| `/tree` | Jump to any point in the session and continue from there |
|
|
54
54
|
| `/fork` | Create a new session from a previous user message |
|
|
55
55
|
| `/clone` | Duplicate the current active branch into a new session |
|
|
@@ -73,6 +73,29 @@ The value is saved as the global `rlmMaxSubagents` preference, so it survives re
|
|
|
73
73
|
|
|
74
74
|
Lowering the limit never kills or passivates existing agents. They keep running or remain idle. Only new spawns/admissions are blocked until the live count is below the limit. This slash command is separate from `base-context agents`, which lists agents.
|
|
75
75
|
|
|
76
|
+
## Token usage and status
|
|
77
|
+
|
|
78
|
+
`/context` (also `/usage`) separates generation usage from the goal budget:
|
|
79
|
+
|
|
80
|
+
- When saved provider-attempt receipts are available, it shows each captured agent's own usage by purpose: main work, child work, compaction, refinement, status, and other recorded generation calls. The captured-family total counts each source once. It does not add assistant or child-attribution projections on top of receipts.
|
|
81
|
+
- Processed tokens equal total input (including cache) plus output. Cached input is not added again. Uncached input, cache read/write, and output are also shown separately.
|
|
82
|
+
- Missing or partial usage, unsettled attempts, unavailable prices, and agents without receipts are explicit. Missing values are not zero. The captured family may not include every historical child or request.
|
|
83
|
+
- Dollar amounts are catalog estimates from recorded rates, not provider invoices. Unavailable prices are not treated as free usage.
|
|
84
|
+
- Older sessions or daemons without receipt data use a labeled legacy conversation projection. It can omit auxiliary calls and is not whole-family provider spend.
|
|
85
|
+
- The goal budget remains **root successful main uncached input + output only**. Cached input, child work, and auxiliary calls do not consume that counter.
|
|
86
|
+
|
|
87
|
+
Dashboard summaries do not request inference again when their bounded input is unchanged. Changed working input and retries are coalesced with a 60-second minimum interval and a short settle debounce. An observed working-to-idle transition can request a final summary without waiting for that interval. Local activity, queues, and terminal errors remain visible without a new summary call. The selected status model is unchanged.
|
|
88
|
+
|
|
89
|
+
## Harness refinement
|
|
90
|
+
|
|
91
|
+
`/refine` updates the editable continual harness. Manual and automatic refinement keep their existing scheduling and scope rules.
|
|
92
|
+
|
|
93
|
+
Static harness instructions stay in the system prompt. Fresh compact entries and selected error-fix advice enter the next owned model request as a saved **Continual Harness Snapshot**, rather than rewriting the system prefix. Direct kernel harness edits use the same path. Unchanged snapshots are not appended again.
|
|
94
|
+
|
|
95
|
+
The latest snapshot supersedes older snapshots, including deletions, rollbacks, and empty state. Entries remain advice below system instructions and the current task. Resume and native context epochs use the retained source records; if compaction removes the snapshot, the next request adds the current state again. Snapshot rows are hidden in the normal UI but remain model-visible.
|
|
96
|
+
|
|
97
|
+
This preserves fresh advice while reducing prefix changes. It does not guarantee a provider cache hit or reduce refinement frequency.
|
|
98
|
+
|
|
76
99
|
## Message Queue
|
|
77
100
|
|
|
78
101
|
You can submit messages while the agent is still working:
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@ponythewhite/base-context",
|
|
3
|
-
"version": "1.0.
|
|
3
|
+
"version": "1.0.13",
|
|
4
4
|
"description": "Synerise base-context: a coding and research agent with durable context and a persistent Python REPL",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"bin": {
|
|
@@ -42,9 +42,9 @@
|
|
|
42
42
|
},
|
|
43
43
|
"dependencies": {
|
|
44
44
|
"@agentclientprotocol/sdk": "^1.3.0",
|
|
45
|
-
"@ponythewhite/base-context-agent": "1.0.
|
|
46
|
-
"@ponythewhite/base-context-ai": "1.0.
|
|
47
|
-
"@ponythewhite/base-context-tui": "1.0.
|
|
45
|
+
"@ponythewhite/base-context-agent": "1.0.13",
|
|
46
|
+
"@ponythewhite/base-context-ai": "1.0.13",
|
|
47
|
+
"@ponythewhite/base-context-tui": "1.0.13",
|
|
48
48
|
"@silvia-odwyer/photon-node": "^0.3.4",
|
|
49
49
|
"chalk": "^5.5.0",
|
|
50
50
|
"cli-highlight": "^2.1.11",
|