@ponythewhite/base-context 1.0.11 → 1.0.13

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (80) hide show
  1. package/CHANGELOG.md +15 -0
  2. package/dist/base-context-runtime/pyproject.toml +1 -1
  3. package/dist/base-context-runtime/uv.lock +1 -1
  4. package/dist/build-info.json +1 -1
  5. package/dist/bundle/amazon-bedrock.js +3 -3
  6. package/dist/bundle/{anthropic-5PSCZEER.js → anthropic-BJVPUMPB.js} +2 -2
  7. package/dist/bundle/{azure-openai-responses-AYTLJR33.js → azure-openai-responses-ERRMAXK6.js} +4 -5
  8. package/dist/bundle/{bundled-modules-PEFT72QN.js → bundled-modules-URKVA3UB.js} +5 -5
  9. package/dist/bundle/{chunk-D3KOF57W.js → chunk-25VYRAC6.js} +14 -1
  10. package/dist/bundle/{chunk-2IPDSTDR.js → chunk-AEMRBMI2.js} +9 -9
  11. package/dist/bundle/{chunk-Q3MBSBEB.js → chunk-FOGKLC7V.js} +1 -1
  12. package/dist/bundle/{chunk-KT2BNEPE.js → chunk-MUFZODP5.js} +1 -1
  13. package/dist/bundle/{chunk-QKIFBA7N.js → chunk-Q6UAHLII.js} +2 -2
  14. package/dist/bundle/{chunk-TOOVREMP.js → chunk-TKOBEJ5E.js} +515 -107
  15. package/dist/bundle/{chunk-JZHGOOFM.js → chunk-TXVF6Q5H.js} +118 -22
  16. package/dist/bundle/{cli-main-BRXGYTGK.js → cli-main-ZTJM5EEH.js} +4 -4
  17. package/dist/bundle/cli.js +1 -1
  18. package/dist/bundle/{google-HTMCUJ7X.js → google-4GA4YMUX.js} +2 -2
  19. package/dist/bundle/{google-vertex-ZQO65AXF.js → google-vertex-7AMALD7D.js} +2 -2
  20. package/dist/bundle/{main-POUXGEL6.js → main-53I74INO.js} +5 -5
  21. package/dist/bundle/{mistral-SB7LU56G.js → mistral-74BXMNRM.js} +1 -1
  22. package/dist/bundle/{openai-codex-responses-5CEW4VN3.js → openai-codex-responses-COXNC4IH.js} +2 -2
  23. package/dist/bundle/{openai-completions-BAZJQSS2.js → openai-completions-BCSA5Q6S.js} +2 -2
  24. package/dist/bundle/{openai-responses-VHOE427P.js → openai-responses-ONCIZSJS.js} +3 -3
  25. package/dist/core/agent-session.d.ts +6 -0
  26. package/dist/core/agent-session.d.ts.map +1 -1
  27. package/dist/core/agent-session.js +362 -103
  28. package/dist/core/agent-session.js.map +1 -1
  29. package/dist/core/compaction/compaction.d.ts +9 -1
  30. package/dist/core/compaction/compaction.d.ts.map +1 -1
  31. package/dist/core/compaction/compaction.js +29 -3
  32. package/dist/core/compaction/compaction.js.map +1 -1
  33. package/dist/core/context-tree.d.ts +13 -10
  34. package/dist/core/context-tree.d.ts.map +1 -1
  35. package/dist/core/context-tree.js +49 -3
  36. package/dist/core/context-tree.js.map +1 -1
  37. package/dist/core/inference-coordinator.d.ts +1 -1
  38. package/dist/core/inference-coordinator.d.ts.map +1 -1
  39. package/dist/core/inference-coordinator.js +18 -5
  40. package/dist/core/inference-coordinator.js.map +1 -1
  41. package/dist/core/messages.d.ts +2 -0
  42. package/dist/core/messages.d.ts.map +1 -1
  43. package/dist/core/messages.js +2 -0
  44. package/dist/core/messages.js.map +1 -1
  45. package/dist/core/refinement/refinement.d.ts +2 -0
  46. package/dist/core/refinement/refinement.d.ts.map +1 -1
  47. package/dist/core/refinement/refinement.js +10 -1
  48. package/dist/core/refinement/refinement.js.map +1 -1
  49. package/dist/core/request-usage.d.ts +34 -0
  50. package/dist/core/request-usage.d.ts.map +1 -0
  51. package/dist/core/request-usage.js +111 -0
  52. package/dist/core/request-usage.js.map +1 -0
  53. package/dist/core/settings-manager.d.ts +3 -0
  54. package/dist/core/settings-manager.d.ts.map +1 -1
  55. package/dist/core/settings-manager.js +7 -0
  56. package/dist/core/settings-manager.js.map +1 -1
  57. package/dist/core/system-prompt.d.ts +2 -0
  58. package/dist/core/system-prompt.d.ts.map +1 -1
  59. package/dist/core/system-prompt.js +4 -2
  60. package/dist/core/system-prompt.js.map +1 -1
  61. package/dist/modes/agent-connection/daemon-agent-connection.d.ts.map +1 -1
  62. package/dist/modes/agent-connection/daemon-agent-connection.js +8 -1
  63. package/dist/modes/agent-connection/daemon-agent-connection.js.map +1 -1
  64. package/dist/modes/daemon/daemon-protocol.d.ts +13 -3
  65. package/dist/modes/daemon/daemon-protocol.d.ts.map +1 -1
  66. package/dist/modes/daemon/daemon-protocol.js +10 -2
  67. package/dist/modes/daemon/daemon-protocol.js.map +1 -1
  68. package/dist/modes/daemon/daemon-session-summarizer.d.ts +3 -0
  69. package/dist/modes/daemon/daemon-session-summarizer.d.ts.map +1 -1
  70. package/dist/modes/daemon/daemon-session-summarizer.js +71 -32
  71. package/dist/modes/daemon/daemon-session-summarizer.js.map +1 -1
  72. package/dist/modes/interactive/components/context-tree-format.d.ts.map +1 -1
  73. package/dist/modes/interactive/components/context-tree-format.js +49 -0
  74. package/dist/modes/interactive/components/context-tree-format.js.map +1 -1
  75. package/docs/compaction.md +57 -4
  76. package/docs/context-management.md +23 -1
  77. package/docs/rlm.md +1 -1
  78. package/docs/settings.md +21 -0
  79. package/docs/usage.md +24 -1
  80. package/package.json +4 -4
@@ -35,16 +35,67 @@ or loss from a summary alone.
35
35
 
36
36
  ### When It Triggers
37
37
 
38
- Auto-compaction triggers when:
38
+ Auto-compaction uses 90% of the model context window as its default soft target:
39
39
 
40
40
  ```
41
- contextTokens > contextWindow - reserveTokens
41
+ softTarget = floor(0.9 * contextWindow)
42
+ threshold = min(contextWindow - reserveTokens, max(softTarget, fixedContextTokens + 4 * keepRecentTokens))
43
+ contextTokens > threshold
42
44
  ```
43
45
 
44
- By default, `reserveTokens` is 16384 tokens (configurable in `~/.base-context/settings.json` or `<project-dir>/.base-context/settings.json`). This leaves room for the LLM's response.
46
+ With default `keepRecentTokens` of 20000 and `reserveTokens` of 16384, before
47
+ fixed-context headroom raises the target, the thresholds are 244800 for a
48
+ 272000-token model, 111616 for a 128000-token model, and 47616 for a 64000-token
49
+ model. The model-window reserve can therefore trigger compaction before 90%.
50
+ The model-window ceiling always applies.
51
+
52
+ `fixedContextTokens` estimates current system instructions, tool schemas, the
53
+ current TaskFrame and latest harness snapshot. It does not count all historical
54
+ messages or obsolete snapshots as fixed. Reserving room above this context avoids
55
+ repeated ineffective summaries when required instructions already exceed the
56
+ soft target. These are local estimates, not exact provider token counts.
57
+
58
+ Set `compaction.targetTokens` to a positive safe integer to replace the soft
59
+ target; required-context headroom can still raise a numeric target. Set
60
+ `"model-limit"` to use exactly `contextWindow - reserveTokens` instead.
61
+ Configure these settings in `~/.base-context/settings.json` or
62
+ `<project-dir>/.base-context/settings.json`.
63
+
64
+ This is a compaction heuristic, not a strict request cap, a validated backend
65
+ limit, or a promise of optimal token use. Earlier summaries trade shorter replay
66
+ for summary calls and possible rereads. Required instructions and replay
67
+ dependencies are not clipped to meet this target. Explicit SDK request-token
68
+ budgets remain separate; see [context management](context-management.md#model-aware-budgets).
45
69
 
46
70
  You can also trigger manually with `/compact [instructions]`, where optional instructions focus the summary — for example `/compact focus on the auth refactor, remember the exact migration command`. The instructions are passed to the summarization prompt with high priority, persisted on the `CompactionEntry`, and shown on the `[compaction]` message in the TUI.
47
71
 
72
+ ### Agent-Requested Compaction
73
+
74
+ The agent can request compaction before the automatic threshold, including when
75
+ using Astra. The bundled `compact` skill is enabled by default and runs from the
76
+ Python REPL:
77
+
78
+ ```python
79
+ await compact.status()
80
+ await compact.run("keep the remaining plan and exact test names")
81
+ ```
82
+
83
+ `status()` reports `tokens`, `context_window`, `percent`, and `scheduled`. Usage
84
+ can be unknown immediately after compaction. `run()` schedules compaction at the
85
+ next turn boundary, not in the middle of the Python cell. It returns
86
+ `{"scheduled": True}` when accepted, or `{"scheduled": False, "reason": ...}`
87
+ when no turn is active or there is nothing to compact yet. Optional instructions
88
+ focus the summary. Repeating the request before the boundary updates them.
89
+ Interrupted tool work can continue after the checkpoint; this does not start new
90
+ work after an otherwise completed turn.
91
+
92
+ This request does not depend on the automatic threshold or `compaction.enabled`.
93
+ It requires context optimization to be on. `compaction.agentCallable: false`
94
+ disables the skill, and disabling Python or bundled skills can remove its normal
95
+ entry point. Each recursive agent uses its own session's compaction path; children
96
+ inherit the parent's compact-skill availability. There is no Astra-specific gate.
97
+ A request compacts the requesting session, not the whole agent tree.
98
+
48
99
  ### How It Works
49
100
 
50
101
  1. **Find cut point**: Walk backwards from newest message, accumulating token estimates until `keepRecentTokens` (default 20k, configurable in `~/.base-context/settings.json` or `<project-dir>/.base-context/settings.json`) is reached
@@ -432,9 +483,11 @@ Configure compaction in `~/.base-context/settings.json` or `<project-dir>/.base-
432
483
  | `enabled` | `true` | Enable auto-compaction |
433
484
  | `reserveTokens` | `16384` | Headroom used by the compaction threshold |
434
485
  | `keepRecentTokens` | `20000` | Estimated recent-token target for the retained tail |
486
+ | `targetTokens` | 90% of the model context window | Positive safe integer to override the soft target, or `"model-limit"` for the model-window threshold; fixed-context headroom and the model reserve still apply as described above |
487
+ | `agentCallable` | `true` | Expose the `compact` skill so the agent can request earlier compaction |
435
488
 
436
489
  Disable automatic compaction with `"compaction": { "enabled": false }`. Manual
437
- `/compact` remains available while `context.mode` is `"on"`. Setting `context.mode`
490
+ `/compact` and agent-requested compaction remain available while `context.mode` is `"on"`. Setting `context.mode`
438
491
  to `"off"` disables context optimization, including manual compaction; logging,
439
492
  recovery, limits and other non-optimization ownership remain active.
440
493
 
@@ -82,7 +82,29 @@ Profiles identify the API, provider, endpoint, final model, context allowance, o
82
82
 
83
83
  In supported native Responses/Codex selection paths, historical assistant literals can become optional at accepted boundaries. Users, the latest assistant, TaskFrames, summaries, recovery, and required dependencies remain mandatory. If that set cannot fit, the request can refuse. Unsupported media, opaque layouts, or missing contracts also remain explicit limits.
84
84
 
85
- See [SDK request-token profiles](sdk.md#explicit-request-token-budget-profiles) for the exact scope and configuration. Ordinary [compaction](compaction.md) and its `reserveTokens`/`keepRecentTokens` settings are separate.
85
+ See [SDK request-token profiles](sdk.md#explicit-request-token-budget-profiles) for the exact scope and configuration.
86
+
87
+ Ordinary [compaction](compaction.md) uses a separate soft target of 90% of the
88
+ model context window: `floor(0.9 * contextWindow)`. The actual trigger also leaves
89
+ `4 * keepRecentTokens` above estimated fixed context, then caps the result at
90
+ `contextWindow - reserveTokens`. Fixed context includes current system
91
+ instructions, tool schemas, the current TaskFrame and latest harness snapshot,
92
+ not every historical message. This avoids repeated ineffective summaries of
93
+ noncompactable instructions.
94
+
95
+ With small fixed context, default thresholds are 244800 estimated tokens for a
96
+ 272000-token model, 111616 for a 128000-token model, and 47616 for a 64000-token
97
+ model. The model-window reserve can trigger compaction before 90%. Set
98
+ `compaction.targetTokens` to a positive safe integer to replace the soft target
99
+ (still subject to fixed-context headroom), or `"model-limit"` to use exactly
100
+ `contextWindow - reserveTokens`. Required replay groups are not clipped to meet
101
+ this target. Summaries can add calls and rereads; this is not a strict request
102
+ cap, an optimality guarantee, or evidence of a deployment's actual limit.
103
+
104
+ The agent can request an earlier checkpoint from the Python REPL with
105
+ `await compact.run()`, independently of the automatic threshold. It can inspect
106
+ usage with `await compact.status()`. This is enabled by default for all models,
107
+ including Astra; see [agent-requested compaction](compaction.md#agent-requested-compaction).
86
108
 
87
109
  ## Stable context epochs
88
110
 
package/docs/rlm.md CHANGED
@@ -101,7 +101,7 @@ await agent_message.send(
101
101
 
102
102
  #### Child handles and lifecycle
103
103
 
104
- An admission handle contains `rlm_child_id`, `name`, `session_dir`, and `model`. Child usage is attributed to the parent session while remaining distinguishable in context-tree reporting.
104
+ An admission handle contains `rlm_child_id`, `name`, `session_dir`, and `model`. Child usage is attributed to the parent session while remaining distinguishable in context-tree reporting. Native child usage reporting continues after passivation and rehydration. `/context` separately totals saved request receipts for the captured family; parent attribution records are not added to those totals. See [token usage and status](usage.md#token-usage-and-status) for missing-data, catalog-estimate, and goal-budget scope.
105
105
 
106
106
  The parent-scoped child registry survives compaction, kernel restart, and parent restoration:
107
107
 
package/docs/settings.md CHANGED
@@ -94,6 +94,8 @@ Private download manifests require `version` and `package` (or `packageName`) se
94
94
  | `compaction.enabled` | boolean | `true` | Enable auto-compaction |
95
95
  | `compaction.reserveTokens` | number | `16384` | Tokens reserved for LLM response |
96
96
  | `compaction.keepRecentTokens` | number | `20000` | Recent tokens to keep (not summarized) |
97
+ | `compaction.targetTokens` | positive integer or `"model-limit"` | 90% of the model context window | Override the soft compaction target, still capped by the model window minus reserve |
98
+ | `compaction.agentCallable` | boolean | `true` | Expose the `compact` skill so the agent can request compaction before the automatic threshold |
97
99
  | `compaction.model` | object | Current main model and effort | Explicit summary model: `provider`, `modelId`, and `thinkingLevel` are all required |
98
100
 
99
101
  ```json
@@ -107,6 +109,25 @@ Private download manifests require `version` and `package` (or `packageName`) se
107
109
  ```
108
110
 
109
111
 
112
+ Without `targetTokens`, the soft target is 90% of the model context window:
113
+ `floor(0.9 * contextWindow)`. The trigger also leaves `4 * keepRecentTokens` above
114
+ an estimate of current fixed instructions, tool schemas, TaskFrame and latest
115
+ harness snapshot. It is capped at `contextWindow - reserveTokens`. With small
116
+ fixed context, default thresholds are 244800 for a 272000-token model, 111616 for a
117
+ 128000-token model, and 47616 for a 64000-token model. The model-window reserve can
118
+ therefore trigger compaction before 90%. A positive safe integer replaces the
119
+ soft target, but required-context headroom can raise it; `"model-limit"` uses
120
+ exactly `contextWindow - reserveTokens`. This is a summary heuristic, not strict
121
+ provider-request admission. See [compaction](compaction.md#when-it-triggers) and
122
+ [model-aware budgets](context-management.md#model-aware-budgets).
123
+
124
+ With the default `compaction.agentCallable: true`, the agent can use
125
+ `await compact.status()` and `await compact.run(instructions=None)` in the Python
126
+ REPL. This model-independent path includes Astra and can request compaction before
127
+ the automatic threshold. It also works with `compaction.enabled: false`, but new
128
+ requests require `context.mode: "on"`. Requests run at a turn boundary, not
129
+ mid-cell. See [agent-requested compaction](compaction.md#agent-requested-compaction).
130
+
110
131
  Set `compaction.model` to choose the model and effort for manual, automatic and
111
132
  model-requested compaction summaries. For example, when this exact model/route is
112
133
  configured and budgeted:
package/docs/usage.md CHANGED
@@ -49,7 +49,7 @@ Type `/` in the editor to open command completion. Extensions can register custo
49
49
  | `/new` | Start a new session |
50
50
  | `/name <name>` | Set session display name |
51
51
  | `/session` | Show session file, ID, and message counts |
52
- | `/usage`, `/context` | Show the parent and subagent context, token, and cost breakdown |
52
+ | `/usage`, `/context` | Show captured-family request usage, context, and separate goal-budget scope |
53
53
  | `/tree` | Jump to any point in the session and continue from there |
54
54
  | `/fork` | Create a new session from a previous user message |
55
55
  | `/clone` | Duplicate the current active branch into a new session |
@@ -73,6 +73,29 @@ The value is saved as the global `rlmMaxSubagents` preference, so it survives re
73
73
 
74
74
  Lowering the limit never kills or passivates existing agents. They keep running or remain idle. Only new spawns/admissions are blocked until the live count is below the limit. This slash command is separate from `base-context agents`, which lists agents.
75
75
 
76
+ ## Token usage and status
77
+
78
+ `/context` (also `/usage`) separates generation usage from the goal budget:
79
+
80
+ - When saved provider-attempt receipts are available, it shows each captured agent's own usage by purpose: main work, child work, compaction, refinement, status, and other recorded generation calls. The captured-family total counts each source once. It does not add assistant or child-attribution projections on top of receipts.
81
+ - Processed tokens equal total input (including cache) plus output. Cached input is not added again. Uncached input, cache read/write, and output are also shown separately.
82
+ - Missing or partial usage, unsettled attempts, unavailable prices, and agents without receipts are explicit. Missing values are not zero. The captured family may not include every historical child or request.
83
+ - Dollar amounts are catalog estimates from recorded rates, not provider invoices. Unavailable prices are not treated as free usage.
84
+ - Older sessions or daemons without receipt data use a labeled legacy conversation projection. It can omit auxiliary calls and is not whole-family provider spend.
85
+ - The goal budget remains **root successful main uncached input + output only**. Cached input, child work, and auxiliary calls do not consume that counter.
86
+
87
+ Dashboard summaries do not request inference again when their bounded input is unchanged. Changed working input and retries are coalesced with a 60-second minimum interval and a short settle debounce. An observed working-to-idle transition can request a final summary without waiting for that interval. Local activity, queues, and terminal errors remain visible without a new summary call. The selected status model is unchanged.
88
+
89
+ ## Harness refinement
90
+
91
+ `/refine` updates the editable continual harness. Manual and automatic refinement keep their existing scheduling and scope rules.
92
+
93
+ Static harness instructions stay in the system prompt. Fresh compact entries and selected error-fix advice enter the next owned model request as a saved **Continual Harness Snapshot**, rather than rewriting the system prefix. Direct kernel harness edits use the same path. Unchanged snapshots are not appended again.
94
+
95
+ The latest snapshot supersedes older snapshots, including deletions, rollbacks, and empty state. Entries remain advice below system instructions and the current task. Resume and native context epochs use the retained source records; if compaction removes the snapshot, the next request adds the current state again. Snapshot rows are hidden in the normal UI but remain model-visible.
96
+
97
+ This preserves fresh advice while reducing prefix changes. It does not guarantee a provider cache hit or reduce refinement frequency.
98
+
76
99
  ## Message Queue
77
100
 
78
101
  You can submit messages while the agent is still working:
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@ponythewhite/base-context",
3
- "version": "1.0.11",
3
+ "version": "1.0.13",
4
4
  "description": "Synerise base-context: a coding and research agent with durable context and a persistent Python REPL",
5
5
  "type": "module",
6
6
  "bin": {
@@ -42,9 +42,9 @@
42
42
  },
43
43
  "dependencies": {
44
44
  "@agentclientprotocol/sdk": "^1.3.0",
45
- "@ponythewhite/base-context-agent": "1.0.11",
46
- "@ponythewhite/base-context-ai": "1.0.11",
47
- "@ponythewhite/base-context-tui": "1.0.11",
45
+ "@ponythewhite/base-context-agent": "1.0.13",
46
+ "@ponythewhite/base-context-ai": "1.0.13",
47
+ "@ponythewhite/base-context-tui": "1.0.13",
48
48
  "@silvia-odwyer/photon-node": "^0.3.4",
49
49
  "chalk": "^5.5.0",
50
50
  "cli-highlight": "^2.1.11",