@ponythewhite/base-context 1.0.11 → 1.0.12

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (80) hide show
  1. package/CHANGELOG.md +11 -0
  2. package/dist/base-context-runtime/pyproject.toml +1 -1
  3. package/dist/base-context-runtime/uv.lock +1 -1
  4. package/dist/build-info.json +1 -1
  5. package/dist/bundle/amazon-bedrock.js +3 -3
  6. package/dist/bundle/{anthropic-5PSCZEER.js → anthropic-BJVPUMPB.js} +2 -2
  7. package/dist/bundle/{azure-openai-responses-AYTLJR33.js → azure-openai-responses-ERRMAXK6.js} +4 -5
  8. package/dist/bundle/{bundled-modules-PEFT72QN.js → bundled-modules-WCGPZI2H.js} +5 -5
  9. package/dist/bundle/{chunk-D3KOF57W.js → chunk-25VYRAC6.js} +14 -1
  10. package/dist/bundle/{chunk-2IPDSTDR.js → chunk-AEMRBMI2.js} +9 -9
  11. package/dist/bundle/{chunk-Q3MBSBEB.js → chunk-FOGKLC7V.js} +1 -1
  12. package/dist/bundle/{chunk-JZHGOOFM.js → chunk-IAKHCQA2.js} +118 -22
  13. package/dist/bundle/{chunk-KT2BNEPE.js → chunk-MUFZODP5.js} +1 -1
  14. package/dist/bundle/{chunk-QKIFBA7N.js → chunk-Q6UAHLII.js} +2 -2
  15. package/dist/bundle/{chunk-TOOVREMP.js → chunk-YGV3Y4AD.js} +515 -107
  16. package/dist/bundle/{cli-main-BRXGYTGK.js → cli-main-CTYDQHFE.js} +4 -4
  17. package/dist/bundle/cli.js +1 -1
  18. package/dist/bundle/{google-HTMCUJ7X.js → google-4GA4YMUX.js} +2 -2
  19. package/dist/bundle/{google-vertex-ZQO65AXF.js → google-vertex-7AMALD7D.js} +2 -2
  20. package/dist/bundle/{main-POUXGEL6.js → main-GVGK354K.js} +5 -5
  21. package/dist/bundle/{mistral-SB7LU56G.js → mistral-74BXMNRM.js} +1 -1
  22. package/dist/bundle/{openai-codex-responses-5CEW4VN3.js → openai-codex-responses-COXNC4IH.js} +2 -2
  23. package/dist/bundle/{openai-completions-BAZJQSS2.js → openai-completions-BCSA5Q6S.js} +2 -2
  24. package/dist/bundle/{openai-responses-VHOE427P.js → openai-responses-ONCIZSJS.js} +3 -3
  25. package/dist/core/agent-session.d.ts +6 -0
  26. package/dist/core/agent-session.d.ts.map +1 -1
  27. package/dist/core/agent-session.js +362 -103
  28. package/dist/core/agent-session.js.map +1 -1
  29. package/dist/core/compaction/compaction.d.ts +9 -1
  30. package/dist/core/compaction/compaction.d.ts.map +1 -1
  31. package/dist/core/compaction/compaction.js +29 -3
  32. package/dist/core/compaction/compaction.js.map +1 -1
  33. package/dist/core/context-tree.d.ts +13 -10
  34. package/dist/core/context-tree.d.ts.map +1 -1
  35. package/dist/core/context-tree.js +49 -3
  36. package/dist/core/context-tree.js.map +1 -1
  37. package/dist/core/inference-coordinator.d.ts +1 -1
  38. package/dist/core/inference-coordinator.d.ts.map +1 -1
  39. package/dist/core/inference-coordinator.js +18 -5
  40. package/dist/core/inference-coordinator.js.map +1 -1
  41. package/dist/core/messages.d.ts +2 -0
  42. package/dist/core/messages.d.ts.map +1 -1
  43. package/dist/core/messages.js +2 -0
  44. package/dist/core/messages.js.map +1 -1
  45. package/dist/core/refinement/refinement.d.ts +2 -0
  46. package/dist/core/refinement/refinement.d.ts.map +1 -1
  47. package/dist/core/refinement/refinement.js +10 -1
  48. package/dist/core/refinement/refinement.js.map +1 -1
  49. package/dist/core/request-usage.d.ts +34 -0
  50. package/dist/core/request-usage.d.ts.map +1 -0
  51. package/dist/core/request-usage.js +111 -0
  52. package/dist/core/request-usage.js.map +1 -0
  53. package/dist/core/settings-manager.d.ts +3 -0
  54. package/dist/core/settings-manager.d.ts.map +1 -1
  55. package/dist/core/settings-manager.js +7 -0
  56. package/dist/core/settings-manager.js.map +1 -1
  57. package/dist/core/system-prompt.d.ts +2 -0
  58. package/dist/core/system-prompt.d.ts.map +1 -1
  59. package/dist/core/system-prompt.js +4 -2
  60. package/dist/core/system-prompt.js.map +1 -1
  61. package/dist/modes/agent-connection/daemon-agent-connection.d.ts.map +1 -1
  62. package/dist/modes/agent-connection/daemon-agent-connection.js +8 -1
  63. package/dist/modes/agent-connection/daemon-agent-connection.js.map +1 -1
  64. package/dist/modes/daemon/daemon-protocol.d.ts +13 -3
  65. package/dist/modes/daemon/daemon-protocol.d.ts.map +1 -1
  66. package/dist/modes/daemon/daemon-protocol.js +10 -2
  67. package/dist/modes/daemon/daemon-protocol.js.map +1 -1
  68. package/dist/modes/daemon/daemon-session-summarizer.d.ts +3 -0
  69. package/dist/modes/daemon/daemon-session-summarizer.d.ts.map +1 -1
  70. package/dist/modes/daemon/daemon-session-summarizer.js +71 -32
  71. package/dist/modes/daemon/daemon-session-summarizer.js.map +1 -1
  72. package/dist/modes/interactive/components/context-tree-format.d.ts.map +1 -1
  73. package/dist/modes/interactive/components/context-tree-format.js +49 -0
  74. package/dist/modes/interactive/components/context-tree-format.js.map +1 -1
  75. package/docs/compaction.md +27 -3
  76. package/docs/context-management.md +17 -1
  77. package/docs/rlm.md +1 -1
  78. package/docs/settings.md +12 -0
  79. package/docs/usage.md +24 -1
  80. package/package.json +4 -4
@@ -35,13 +35,36 @@ or loss from a summary alone.
35
35
 
36
36
  ### When It Triggers
37
37
 
38
- Auto-compaction triggers when:
38
+ Auto-compaction uses an earlier, model-aware soft target by default:
39
39
 
40
40
  ```
41
- contextTokens > contextWindow - reserveTokens
41
+ softTarget = max(4 * keepRecentTokens, min(96000, contextWindow / 2))
42
+ threshold = min(contextWindow - reserveTokens, max(softTarget, fixedContextTokens + 4 * keepRecentTokens))
43
+ contextTokens > threshold
42
44
  ```
43
45
 
44
- By default, `reserveTokens` is 16384 tokens (configurable in `~/.base-context/settings.json` or `<project-dir>/.base-context/settings.json`). This leaves room for the LLM's response.
46
+ With default `keepRecentTokens` of 20000 and `reserveTokens` of 16384, before
47
+ fixed-context headroom raises the target, the thresholds are 96000 for a
48
+ 272000-token model, 80000 for a 128000-token model, and 47616 for a 64000-token
49
+ model. The model-window ceiling always applies.
50
+
51
+ `fixedContextTokens` estimates current system instructions, tool schemas, the
52
+ current TaskFrame and latest harness snapshot. It does not count all historical
53
+ messages or obsolete snapshots as fixed. Reserving room above this context avoids
54
+ repeated ineffective summaries when required instructions already exceed the
55
+ soft target. These are local estimates, not exact provider token counts.
56
+
57
+ Set `compaction.targetTokens` to a positive safe integer to replace the soft
58
+ target; required-context headroom can still raise a numeric target. Set
59
+ `"model-limit"` to use exactly `contextWindow - reserveTokens` instead.
60
+ Configure these settings in `~/.base-context/settings.json` or
61
+ `<project-dir>/.base-context/settings.json`.
62
+
63
+ This is a compaction heuristic, not a strict request cap, a validated backend
64
+ limit, or a promise of optimal token use. Earlier summaries trade shorter replay
65
+ for summary calls and possible rereads. Required instructions and replay
66
+ dependencies are not clipped to meet this target. Explicit SDK request-token
67
+ budgets remain separate; see [context management](context-management.md#model-aware-budgets).
45
68
 
46
69
  You can also trigger manually with `/compact [instructions]`, where optional instructions focus the summary — for example `/compact focus on the auth refactor, remember the exact migration command`. The instructions are passed to the summarization prompt with high priority, persisted on the `CompactionEntry`, and shown on the `[compaction]` message in the TUI.
47
70
 
@@ -432,6 +455,7 @@ Configure compaction in `~/.base-context/settings.json` or `<project-dir>/.base-
432
455
  | `enabled` | `true` | Enable auto-compaction |
433
456
  | `reserveTokens` | `16384` | Headroom used by the compaction threshold |
434
457
  | `keepRecentTokens` | `20000` | Estimated recent-token target for the retained tail |
458
+ | `targetTokens` | Model-aware soft target | Positive safe integer to override the soft target, or `"model-limit"` for the model-window threshold |
435
459
 
436
460
  Disable automatic compaction with `"compaction": { "enabled": false }`. Manual
437
461
  `/compact` remains available while `context.mode` is `"on"`. Setting `context.mode`
@@ -82,7 +82,23 @@ Profiles identify the API, provider, endpoint, final model, context allowance, o
82
82
 
83
83
  In supported native Responses/Codex selection paths, historical assistant literals can become optional at accepted boundaries. Users, the latest assistant, TaskFrames, summaries, recovery, and required dependencies remain mandatory. If that set cannot fit, the request can refuse. Unsupported media, opaque layouts, or missing contracts also remain explicit limits.
84
84
 
85
- See [SDK request-token profiles](sdk.md#explicit-request-token-budget-profiles) for the exact scope and configuration. Ordinary [compaction](compaction.md) and its `reserveTokens`/`keepRecentTokens` settings are separate.
85
+ See [SDK request-token profiles](sdk.md#explicit-request-token-budget-profiles) for the exact scope and configuration.
86
+
87
+ Ordinary [compaction](compaction.md) uses a separate model-aware soft target:
88
+ `max(4 * keepRecentTokens, min(96000, contextWindow / 2))`. The actual trigger also
89
+ leaves `4 * keepRecentTokens` above estimated fixed context, then caps the result
90
+ at `contextWindow - reserveTokens`. Fixed context includes current system
91
+ instructions, tool schemas, the current TaskFrame and latest harness snapshot,
92
+ not every historical message. This avoids repeated ineffective summaries of
93
+ noncompactable instructions.
94
+
95
+ With small fixed context, default settings trigger earlier at 96000 estimated
96
+ tokens for a 272000-token model. Set `compaction.targetTokens` to a positive safe
97
+ integer to replace the soft target (still subject to fixed-context headroom), or
98
+ `"model-limit"` to use exactly the full-window threshold.
99
+ This policy reduces repeated large-history requests without clipping required
100
+ replay groups. It can add summary calls and rereads; it is not a strict request
101
+ cap, an optimality guarantee, or evidence of a deployment's actual limit.
86
102
 
87
103
  ## Stable context epochs
88
104
 
package/docs/rlm.md CHANGED
@@ -101,7 +101,7 @@ await agent_message.send(
101
101
 
102
102
  #### Child handles and lifecycle
103
103
 
104
- An admission handle contains `rlm_child_id`, `name`, `session_dir`, and `model`. Child usage is attributed to the parent session while remaining distinguishable in context-tree reporting.
104
+ An admission handle contains `rlm_child_id`, `name`, `session_dir`, and `model`. Child usage is attributed to the parent session while remaining distinguishable in context-tree reporting. Native child usage reporting continues after passivation and rehydration. `/context` separately totals saved request receipts for the captured family; parent attribution records are not added to those totals. See [token usage and status](usage.md#token-usage-and-status) for missing-data, catalog-estimate, and goal-budget scope.
105
105
 
106
106
  The parent-scoped child registry survives compaction, kernel restart, and parent restoration:
107
107
 
package/docs/settings.md CHANGED
@@ -94,6 +94,7 @@ Private download manifests require `version` and `package` (or `packageName`) se
94
94
  | `compaction.enabled` | boolean | `true` | Enable auto-compaction |
95
95
  | `compaction.reserveTokens` | number | `16384` | Tokens reserved for LLM response |
96
96
  | `compaction.keepRecentTokens` | number | `20000` | Recent tokens to keep (not summarized) |
97
+ | `compaction.targetTokens` | positive integer or `"model-limit"` | Model-aware soft target | Override the soft compaction target, still capped by the model window minus reserve |
97
98
  | `compaction.model` | object | Current main model and effort | Explicit summary model: `provider`, `modelId`, and `thinkingLevel` are all required |
98
99
 
99
100
  ```json
@@ -107,6 +108,17 @@ Private download manifests require `version` and `package` (or `packageName`) se
107
108
  ```
108
109
 
109
110
 
111
+ Without `targetTokens`, the soft target is
112
+ `max(4 * keepRecentTokens, min(96000, contextWindow / 2))`. The trigger also leaves
113
+ `4 * keepRecentTokens` above an estimate of current fixed instructions, tool
114
+ schemas, TaskFrame and latest harness snapshot. It is capped at
115
+ `contextWindow - reserveTokens`. With small fixed context, default settings use
116
+ 96000 for a 272000-token model. A positive safe integer replaces the soft target,
117
+ but required-context headroom can raise it; `"model-limit"` uses exactly the
118
+ full-window threshold. This is a summary heuristic,
119
+ not strict provider-request admission. See [compaction](compaction.md#when-it-triggers)
120
+ and [model-aware budgets](context-management.md#model-aware-budgets).
121
+
110
122
  Set `compaction.model` to choose the model and effort for manual, automatic and
111
123
  model-requested compaction summaries. For example, when this exact model/route is
112
124
  configured and budgeted:
package/docs/usage.md CHANGED
@@ -49,7 +49,7 @@ Type `/` in the editor to open command completion. Extensions can register custo
49
49
  | `/new` | Start a new session |
50
50
  | `/name <name>` | Set session display name |
51
51
  | `/session` | Show session file, ID, and message counts |
52
- | `/usage`, `/context` | Show the parent and subagent context, token, and cost breakdown |
52
+ | `/usage`, `/context` | Show captured-family request usage, context, and separate goal-budget scope |
53
53
  | `/tree` | Jump to any point in the session and continue from there |
54
54
  | `/fork` | Create a new session from a previous user message |
55
55
  | `/clone` | Duplicate the current active branch into a new session |
@@ -73,6 +73,29 @@ The value is saved as the global `rlmMaxSubagents` preference, so it survives re
73
73
 
74
74
  Lowering the limit never kills or passivates existing agents. They keep running or remain idle. Only new spawns/admissions are blocked until the live count is below the limit. This slash command is separate from `base-context agents`, which lists agents.
75
75
 
76
+ ## Token usage and status
77
+
78
+ `/context` (also `/usage`) separates generation usage from the goal budget:
79
+
80
+ - When saved provider-attempt receipts are available, it shows each captured agent's own usage by purpose: main work, child work, compaction, refinement, status, and other recorded generation calls. The captured-family total counts each source once. It does not add assistant or child-attribution projections on top of receipts.
81
+ - Processed tokens equal total input (including cache) plus output. Cached input is not added again. Uncached input, cache read/write, and output are also shown separately.
82
+ - Missing or partial usage, unsettled attempts, unavailable prices, and agents without receipts are explicit. Missing values are not zero. The captured family may not include every historical child or request.
83
+ - Dollar amounts are catalog estimates from recorded rates, not provider invoices. Unavailable prices are not treated as free usage.
84
+ - Older sessions or daemons without receipt data use a labeled legacy conversation projection. It can omit auxiliary calls and is not whole-family provider spend.
85
+ - The goal budget remains **root successful main uncached input + output only**. Cached input, child work, and auxiliary calls do not consume that counter.
86
+
87
+ Dashboard summaries do not request inference again when their bounded input is unchanged. Changed working input and retries are coalesced with a 60-second minimum interval and a short settle debounce. An observed working-to-idle transition can request a final summary without waiting for that interval. Local activity, queues, and terminal errors remain visible without a new summary call. The selected status model is unchanged.
88
+
89
+ ## Harness refinement
90
+
91
+ `/refine` updates the editable continual harness. Manual and automatic refinement keep their existing scheduling and scope rules.
92
+
93
+ Static harness instructions stay in the system prompt. Fresh compact entries and selected error-fix advice enter the next owned model request as a saved **Continual Harness Snapshot**, rather than rewriting the system prefix. Direct kernel harness edits use the same path. Unchanged snapshots are not appended again.
94
+
95
+ The latest snapshot supersedes older snapshots, including deletions, rollbacks, and empty state. Entries remain advice below system instructions and the current task. Resume and native context epochs use the retained source records; if compaction removes the snapshot, the next request adds the current state again. Snapshot rows are hidden in the normal UI but remain model-visible.
96
+
97
+ This preserves fresh advice while reducing prefix changes. It does not guarantee a provider cache hit or reduce refinement frequency.
98
+
76
99
  ## Message Queue
77
100
 
78
101
  You can submit messages while the agent is still working:
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@ponythewhite/base-context",
3
- "version": "1.0.11",
3
+ "version": "1.0.12",
4
4
  "description": "Synerise base-context: a coding and research agent with durable context and a persistent Python REPL",
5
5
  "type": "module",
6
6
  "bin": {
@@ -42,9 +42,9 @@
42
42
  },
43
43
  "dependencies": {
44
44
  "@agentclientprotocol/sdk": "^1.3.0",
45
- "@ponythewhite/base-context-agent": "1.0.11",
46
- "@ponythewhite/base-context-ai": "1.0.11",
47
- "@ponythewhite/base-context-tui": "1.0.11",
45
+ "@ponythewhite/base-context-agent": "1.0.12",
46
+ "@ponythewhite/base-context-ai": "1.0.12",
47
+ "@ponythewhite/base-context-tui": "1.0.12",
48
48
  "@silvia-odwyer/photon-node": "^0.3.4",
49
49
  "chalk": "^5.5.0",
50
50
  "cli-highlight": "^2.1.11",