@ashlr/hub 3.8.0 → 3.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (147) hide show
  1. package/CHANGELOG.md +201 -0
  2. package/dist/build-identity.json +2 -2
  3. package/dist/cli/resource-profile.js +95 -21
  4. package/dist/cli/resource-profile.js.map +1 -1
  5. package/dist/core/resources/native-profile.d.ts +56 -0
  6. package/dist/core/resources/native-profile.js +368 -6
  7. package/dist/core/resources/native-profile.js.map +1 -1
  8. package/dist/core/run/model-catalog.js +8 -1
  9. package/dist/core/run/model-catalog.js.map +1 -1
  10. package/dist/core/universe/builtins/preparation/manifest.json +1 -1
  11. package/dist/core/universe/builtins/preparation/preparation-bridge.mjs +9 -2
  12. package/dist/core/universe/builtins/preparation/preparation-verification-fixtures.mjs +8 -1
  13. package/dist/core/verse/adapters/claude.d.ts +49 -2
  14. package/dist/core/verse/adapters/claude.js +317 -56
  15. package/dist/core/verse/adapters/claude.js.map +1 -1
  16. package/dist/core/verse/adapters/codex.d.ts +81 -5
  17. package/dist/core/verse/adapters/codex.js +409 -47
  18. package/dist/core/verse/adapters/codex.js.map +1 -1
  19. package/dist/core/verse/adapters/grok.d.ts +13 -1
  20. package/dist/core/verse/adapters/grok.js +28 -2
  21. package/dist/core/verse/adapters/grok.js.map +1 -1
  22. package/dist/core/verse/adapters/index.d.ts +32 -0
  23. package/dist/core/verse/adapters/index.js +2 -0
  24. package/dist/core/verse/adapters/index.js.map +1 -1
  25. package/dist/core/verse/codex-rollout.d.ts +284 -0
  26. package/dist/core/verse/codex-rollout.js +786 -0
  27. package/dist/core/verse/codex-rollout.js.map +1 -0
  28. package/dist/core/verse/context-fit.d.ts +63 -0
  29. package/dist/core/verse/context-fit.js +355 -0
  30. package/dist/core/verse/context-fit.js.map +1 -0
  31. package/dist/core/verse/context-math.d.ts +215 -0
  32. package/dist/core/verse/context-math.js +365 -0
  33. package/dist/core/verse/context-math.js.map +1 -0
  34. package/dist/core/verse/local-models.d.ts +137 -1
  35. package/dist/core/verse/local-models.js +209 -5
  36. package/dist/core/verse/local-models.js.map +1 -1
  37. package/dist/core/verse/model-windows.d.ts +217 -0
  38. package/dist/core/verse/model-windows.js +616 -0
  39. package/dist/core/verse/model-windows.js.map +1 -0
  40. package/dist/core/verse/preferences.d.ts +123 -0
  41. package/dist/core/verse/preferences.js +403 -0
  42. package/dist/core/verse/preferences.js.map +1 -0
  43. package/dist/core/verse/project-memory.d.ts +112 -0
  44. package/dist/core/verse/project-memory.js +295 -0
  45. package/dist/core/verse/project-memory.js.map +1 -0
  46. package/dist/core/verse/seats.d.ts +106 -4
  47. package/dist/core/verse/seats.js +291 -89
  48. package/dist/core/verse/seats.js.map +1 -1
  49. package/dist/core/verse/session-engine.d.ts +76 -2
  50. package/dist/core/verse/session-engine.js +684 -63
  51. package/dist/core/verse/session-engine.js.map +1 -1
  52. package/dist/core/verse/session-handoff.d.ts +94 -0
  53. package/dist/core/verse/session-handoff.js +551 -0
  54. package/dist/core/verse/session-handoff.js.map +1 -0
  55. package/dist/core/verse/session-search.d.ts +63 -0
  56. package/dist/core/verse/session-search.js +213 -0
  57. package/dist/core/verse/session-search.js.map +1 -0
  58. package/dist/core/verse/session-store.d.ts +16 -2
  59. package/dist/core/verse/session-store.js +80 -6
  60. package/dist/core/verse/session-store.js.map +1 -1
  61. package/dist/core/verse/types.d.ts +333 -1
  62. package/dist/core/verse/types.js +54 -2
  63. package/dist/core/verse/types.js.map +1 -1
  64. package/dist/core/verse/verse-api.d.ts +40 -5
  65. package/dist/core/verse/verse-api.js +724 -25
  66. package/dist/core/verse/verse-api.js.map +1 -1
  67. package/dist/core/web/api.d.ts +14 -4
  68. package/dist/core/web/api.js +26 -11
  69. package/dist/core/web/api.js.map +1 -1
  70. package/dist/core/web/public/next/assets/{App-BsbJFA2j.js → App-XURmiAAp.js} +5 -5
  71. package/dist/core/web/public/next/assets/{App-BsbJFA2j.js.map → App-XURmiAAp.js.map} +1 -1
  72. package/dist/core/web/public/next/assets/{ApprovalsSection-2CCSoBTJ.js → ApprovalsSection-BriUzoZc.js} +2 -2
  73. package/dist/core/web/public/next/assets/{ApprovalsSection-2CCSoBTJ.js.map → ApprovalsSection-BriUzoZc.js.map} +1 -1
  74. package/dist/core/web/public/next/assets/AutonomySection-C4GBsQVc.js +2 -0
  75. package/dist/core/web/public/next/assets/{AutonomySection-yeNNUPur.js.map → AutonomySection-C4GBsQVc.js.map} +1 -1
  76. package/dist/core/web/public/next/assets/ChatSection-BpOBOUvg.css +1 -0
  77. package/dist/core/web/public/next/assets/ChatSection-hlgPBRco.js +84 -0
  78. package/dist/core/web/public/next/assets/ChatSection-hlgPBRco.js.map +1 -0
  79. package/dist/core/web/public/next/assets/ConfirmDialog-BdQYI-Ng.js +2 -0
  80. package/dist/core/web/public/next/assets/ConfirmDialog-BdQYI-Ng.js.map +1 -0
  81. package/dist/core/web/public/next/assets/{Dialog-nK15L1M0.js → Dialog-BxdIfmFH.js} +2 -2
  82. package/dist/core/web/public/next/assets/{Dialog-nK15L1M0.js.map → Dialog-BxdIfmFH.js.map} +1 -1
  83. package/dist/core/web/public/next/assets/{DiffViewer-D369eZe8.js → DiffViewer-DmeRWuc4.js} +2 -2
  84. package/dist/core/web/public/next/assets/{DiffViewer-D369eZe8.js.map → DiffViewer-DmeRWuc4.js.map} +1 -1
  85. package/dist/core/web/public/next/assets/{McpSection-DVgiSYG1.js → McpSection-q34T7Tho.js} +2 -2
  86. package/dist/core/web/public/next/assets/{McpSection-DVgiSYG1.js.map → McpSection-q34T7Tho.js.map} +1 -1
  87. package/dist/core/web/public/next/assets/{MutationTokenDialog-B9I4PAcp.js → MutationTokenDialog-CFAkwOEt.js} +2 -2
  88. package/dist/core/web/public/next/assets/{MutationTokenDialog-B9I4PAcp.js.map → MutationTokenDialog-CFAkwOEt.js.map} +1 -1
  89. package/dist/core/web/public/next/assets/{ResourcePoolConsoleApp-z6tJj3s9.js → ResourcePoolConsoleApp-DHc8GFOY.js} +4 -4
  90. package/dist/core/web/public/next/assets/{ResourcePoolConsoleApp-z6tJj3s9.js.map → ResourcePoolConsoleApp-DHc8GFOY.js.map} +1 -1
  91. package/dist/core/web/public/next/assets/{RouteErrorBoundary-OSajL45F.js → RouteErrorBoundary-ZL9F4qUn.js} +2 -2
  92. package/dist/core/web/public/next/assets/{RouteErrorBoundary-OSajL45F.js.map → RouteErrorBoundary-ZL9F4qUn.js.map} +1 -1
  93. package/dist/core/web/public/next/assets/{Meter-CLcr-4wS.css → Segmented-8cf0Fypn.css} +1 -1
  94. package/dist/core/web/public/next/assets/{Meter-Cmo1pLbk.js → Segmented-DbRjVye3.js} +2 -2
  95. package/dist/core/web/public/next/assets/Segmented-DbRjVye3.js.map +1 -0
  96. package/dist/core/web/public/next/assets/SettingsSection-nJGGGYqv.js +2 -0
  97. package/dist/core/web/public/next/assets/{SettingsSection-B3YH2Npi.js.map → SettingsSection-nJGGGYqv.js.map} +1 -1
  98. package/dist/core/web/public/next/assets/UniverseConsoleApp-6GJydgo4.js +2 -0
  99. package/dist/core/web/public/next/assets/{UniverseConsoleApp-wEKIzl1M.js.map → UniverseConsoleApp-6GJydgo4.js.map} +1 -1
  100. package/dist/core/web/public/next/assets/{UniverseView-CQVpJqqI.js → UniverseView-DLDtNYL8.js} +2 -2
  101. package/dist/core/web/public/next/assets/{UniverseView-CQVpJqqI.js.map → UniverseView-DLDtNYL8.js.map} +1 -1
  102. package/dist/core/web/public/next/assets/UsageSection-Cf9d9iFH.js +2 -0
  103. package/dist/core/web/public/next/assets/UsageSection-Cf9d9iFH.js.map +1 -0
  104. package/dist/core/web/public/next/assets/VerseConsoleApp-DsdKQoTH.css +1 -0
  105. package/dist/core/web/public/next/assets/VerseConsoleApp-y0uOHixd.js +3 -0
  106. package/dist/core/web/public/next/assets/VerseConsoleApp-y0uOHixd.js.map +1 -0
  107. package/dist/core/web/public/next/assets/{charts-BopViIha.js → charts-CVooqJWN.js} +2 -2
  108. package/dist/core/web/public/next/assets/{charts-BopViIha.js.map → charts-CVooqJWN.js.map} +1 -1
  109. package/dist/core/web/public/next/assets/context-model-Bqv8DVuH.js +2 -0
  110. package/dist/core/web/public/next/assets/context-model-Bqv8DVuH.js.map +1 -0
  111. package/dist/core/web/public/next/assets/global-eaY9GyX3.js +2 -0
  112. package/dist/core/web/public/next/assets/global-eaY9GyX3.js.map +1 -0
  113. package/dist/core/web/public/next/assets/{icons-L06pQH6j.js → icons-vex8rclC.js} +2 -2
  114. package/dist/core/web/public/next/assets/{icons-L06pQH6j.js.map → icons-vex8rclC.js.map} +1 -1
  115. package/dist/core/web/public/next/assets/{index-CXgVQHek.js → index-D2Eik-zt.js} +3 -3
  116. package/dist/core/web/public/next/assets/{index-CXgVQHek.js.map → index-D2Eik-zt.js.map} +1 -1
  117. package/dist/core/web/public/next/assets/{queries-D3sSlan1.js → queries-bAYU0VNy.js} +2 -2
  118. package/dist/core/web/public/next/assets/{queries-D3sSlan1.js.map → queries-bAYU0VNy.js.map} +1 -1
  119. package/dist/core/web/public/next/assets/use-guarded-action-C7QrP7Yp.js +2 -0
  120. package/dist/core/web/public/next/assets/{use-guarded-action-BxfJ9E-n.js.map → use-guarded-action-C7QrP7Yp.js.map} +1 -1
  121. package/dist/core/web/public/next/assets/verse-model-TfDk9cky.js +2 -0
  122. package/dist/core/web/public/next/assets/verse-model-TfDk9cky.js.map +1 -0
  123. package/dist/core/web/public/next/assets/{workspace-model-v0a1gl-C.js → workspace-model-CHol5RVp.js} +2 -2
  124. package/dist/core/web/public/next/assets/{workspace-model-v0a1gl-C.js.map → workspace-model-CHol5RVp.js.map} +1 -1
  125. package/dist/core/web/public/next/index.html +1 -1
  126. package/dist/release-dependency-inventory.json +1 -1
  127. package/docs/README.md +1 -0
  128. package/package.json +1 -1
  129. package/dist/core/web/public/next/assets/AutonomySection-yeNNUPur.js +0 -2
  130. package/dist/core/web/public/next/assets/ChatSection-BfnLe5dT.css +0 -1
  131. package/dist/core/web/public/next/assets/ChatSection-DwDDnCZQ.js +0 -80
  132. package/dist/core/web/public/next/assets/ChatSection-DwDDnCZQ.js.map +0 -1
  133. package/dist/core/web/public/next/assets/ConfirmDialog-C363ty_c.js +0 -2
  134. package/dist/core/web/public/next/assets/ConfirmDialog-C363ty_c.js.map +0 -1
  135. package/dist/core/web/public/next/assets/Meter-Cmo1pLbk.js.map +0 -1
  136. package/dist/core/web/public/next/assets/SettingsSection-B3YH2Npi.js +0 -2
  137. package/dist/core/web/public/next/assets/UniverseConsoleApp-wEKIzl1M.js +0 -2
  138. package/dist/core/web/public/next/assets/UsageSection-dToBThXe.js +0 -2
  139. package/dist/core/web/public/next/assets/UsageSection-dToBThXe.js.map +0 -1
  140. package/dist/core/web/public/next/assets/VerseConsoleApp-CMkCizvL.css +0 -1
  141. package/dist/core/web/public/next/assets/VerseConsoleApp-DHU32JCQ.js +0 -3
  142. package/dist/core/web/public/next/assets/VerseConsoleApp-DHU32JCQ.js.map +0 -1
  143. package/dist/core/web/public/next/assets/global-DMWakEhP.js +0 -2
  144. package/dist/core/web/public/next/assets/global-DMWakEhP.js.map +0 -1
  145. package/dist/core/web/public/next/assets/use-guarded-action-BxfJ9E-n.js +0 -2
  146. package/dist/core/web/public/next/assets/verse-model-CvtK6vP6.js +0 -2
  147. package/dist/core/web/public/next/assets/verse-model-CvtK6vP6.js.map +0 -1
@@ -1951,13 +1951,20 @@ var init_model_catalog = __esm({
1951
1951
  // minEffort 2: measured ~590 s/task against 53-381 s for the previous
1952
1952
  // generation. Consistent, but too slow to be the pick for trivial work —
1953
1953
  // `maxEffort: 1` queries keep falling through to the lighter entries.
1954
+ //
1955
+ // NO 'long-context' (Verse 3.9). The tag is `…-ctx64k`: its Modelfile pins
1956
+ // `num_ctx 65536`, Ollama serves it at exactly 65536 (`n_ctx_slot = 65536`
1957
+ // in every server log), and on the llama-server lane each slot gets 65536
1958
+ // too. That is below the >=100k the capability promises, so tagging it sent
1959
+ // long-context work to a model that would overflow. The architecture's
1960
+ // 262144 is reachable only through a differently pinned tag.
1954
1961
  {
1955
1962
  id: `local-coder:${DEFAULT_LOCAL_MODEL_TAG}`,
1956
1963
  engine: "local-coder",
1957
1964
  tier: "mid",
1958
1965
  costPerMTokIn: 0,
1959
1966
  costPerMTokOut: 0,
1960
- capabilities: ["coder", "reasoning", "long-context"],
1967
+ capabilities: ["coder", "reasoning"],
1961
1968
  minEffort: 2
1962
1969
  },
1963
1970
  // -- Local -- small (catch-all tiny model for trivial tasks) --------------
@@ -11,6 +11,25 @@
11
11
  * llama-server lane — the normalising proxy in front of llama-server. The
12
12
  * adapter does not choose; it forwards the choice `seats.ts` already made.
13
13
  *
14
+ * V3.9 context flags (docs/VERSE-CONTEXT.md). Every flag and env var below was
15
+ * checked on BOTH binaries a seat can be pinned to — `claude --help` on
16
+ * 2.1.257 and 2.1.280 lists `--autocompact <auto|tokens>`,
17
+ * `--append-system-prompt <prompt>`, `--add-dir <directories...>` and
18
+ * `--exclude-dynamic-system-prompt-sections`, and both binaries' strings
19
+ * contain `CLAUDE_CODE_MAX_CONTEXT_TOKENS`:
20
+ *
21
+ * - claude seats, 1M-native models: `--autocompact 400000` in standard mode,
22
+ * `--autocompact auto` in expansive (context-math `claudeAutocompactFlag`).
23
+ * 200k models get no flag — they already compact near 167k natively.
24
+ * - local seats: `CLAUDE_CODE_MAX_CONTEXT_TOKENS=<the session's window>` so
25
+ * the CLI compacts before Ollama truncates (without it the CLI assumes its
26
+ * 200k unknown-model default and never compacts a 64k runner), plus
27
+ * `--exclude-dynamic-system-prompt-sections` so the per-launch git status
28
+ * and env block stop invalidating the local runner's prefix cache.
29
+ * - shared project memory (launch.memory): `--add-dir <dir>` when writable and
30
+ * `--append-system-prompt=<block>`, the SAME snapshotted block every turn so
31
+ * the prompt prefix stays byte-identical.
32
+ *
14
33
  * Parse: claude's stream-json is JSONL where streaming deltas are wrapped as
15
34
  * `{type:'stream_event', event:{...Anthropic Messages wire event}}` and whole
16
35
  * messages arrive as `{type:'assistant'|'user', message:{content:[...]}}`,
@@ -21,17 +40,45 @@ import type { VerseAdapter, VerseTurnParser } from './index.js';
21
40
  type JsonObject = Record<string, unknown>;
22
41
  /** Parse one JSONL line into an object, or null for anything that is not one. */
23
42
  export declare function parseJsonObjectLine(line: string): JsonObject | null;
43
+ /**
44
+ * The context window the CLI reported for this turn's model in
45
+ * `result.modelUsage` — the same number it compacts against, computed at
46
+ * result time, so it already reflects a long-context credit clamp (1M → 200k)
47
+ * or grok's mid-session window upgrade.
48
+ *
49
+ * Which row: `modelUsage` is keyed per model and a turn can touch several
50
+ * (claude runs side calls on Haiku). So, in order:
51
+ * 1. the row whose key IS the turn's model (canonicalised — a record naming
52
+ * the retired alias `claude-opus-5.5` still finds `claude-opus-5-5`);
53
+ * 2. the same comparison with a `[1m]` suffix stripped from both sides;
54
+ * 3. the ONLY row carrying a numeric `contextWindow` — grok's key can differ
55
+ * from the CLI id (`grok-4.6-build` for `grok-4.6`) and grok puts the
56
+ * window on the current model's row alone.
57
+ * Anything else (several windowed rows, none matching) is ambiguous and yields
58
+ * null: an unknown window is shown as unknown, never guessed.
59
+ */
60
+ export declare function runtimeContextWindow(modelUsage: unknown, model: string | null): number | null;
24
61
  /**
25
62
  * Shared parser for the Anthropic Messages wire format. Accepts:
26
63
  * - bare wire events (`message_start`, `content_block_start`, `content_block_delta`,
27
64
  * `content_block_stop`, `message_delta`, `message_stop`) — grok's streaming-messages-json;
28
65
  * - the same events wrapped in claude's `{type:'stream_event', event:{...}}`;
29
- * - claude's whole-message envelopes `assistant` / `user` and the terminal `result`;
30
- * - `system` (`init` captures session_id).
66
+ * - claude's whole-message envelopes `assistant` / `user` and the terminal `result`
67
+ * (whose `modelUsage` carries the CLI's context window for the turn);
68
+ * - `system` (`init` captures session_id and model; `compact_boundary`
69
+ * becomes a `compaction` event).
31
70
  *
32
71
  * A text/tool_use/thinking block can arrive twice — once via the streamed
33
72
  * content_block_* events and again inside an `assistant` envelope — so block
34
73
  * events are deduplicated by content within a turn.
74
+ *
75
+ * Usage is deduplicated the same way, by API call. With partial messages on,
76
+ * ONE call is reported by `message_start` + `message_delta` AND by one
77
+ * `assistant` envelope per content block. Summing every report (as this parser
78
+ * once did) counted a two-block call three times; the turn total only looked
79
+ * right because the terminal `result` replaced it. Each call is keyed by its
80
+ * message id and its readings are merged, so the fallback total is right too
81
+ * when a process dies before `result`.
35
82
  */
36
83
  /**
37
84
  * Shared by the claude/local and grok adapters — both speak the Anthropic wire
@@ -11,13 +11,34 @@
11
11
  * llama-server lane — the normalising proxy in front of llama-server. The
12
12
  * adapter does not choose; it forwards the choice `seats.ts` already made.
13
13
  *
14
+ * V3.9 context flags (docs/VERSE-CONTEXT.md). Every flag and env var below was
15
+ * checked on BOTH binaries a seat can be pinned to — `claude --help` on
16
+ * 2.1.257 and 2.1.280 lists `--autocompact <auto|tokens>`,
17
+ * `--append-system-prompt <prompt>`, `--add-dir <directories...>` and
18
+ * `--exclude-dynamic-system-prompt-sections`, and both binaries' strings
19
+ * contain `CLAUDE_CODE_MAX_CONTEXT_TOKENS`:
20
+ *
21
+ * - claude seats, 1M-native models: `--autocompact 400000` in standard mode,
22
+ * `--autocompact auto` in expansive (context-math `claudeAutocompactFlag`).
23
+ * 200k models get no flag — they already compact near 167k natively.
24
+ * - local seats: `CLAUDE_CODE_MAX_CONTEXT_TOKENS=<the session's window>` so
25
+ * the CLI compacts before Ollama truncates (without it the CLI assumes its
26
+ * 200k unknown-model default and never compacts a 64k runner), plus
27
+ * `--exclude-dynamic-system-prompt-sections` so the per-launch git status
28
+ * and env block stop invalidating the local runner's prefix cache.
29
+ * - shared project memory (launch.memory): `--add-dir <dir>` when writable and
30
+ * `--append-system-prompt=<block>`, the SAME snapshotted block every turn so
31
+ * the prompt prefix stays byte-identical.
32
+ *
14
33
  * Parse: claude's stream-json is JSONL where streaming deltas are wrapped as
15
34
  * `{type:'stream_event', event:{...Anthropic Messages wire event}}` and whole
16
35
  * messages arrive as `{type:'assistant'|'user', message:{content:[...]}}`,
17
36
  * followed by `{type:'result', ...}`. The wire-format machinery is shared with
18
37
  * the grok adapter, which emits the same events without the wrapper.
19
38
  */
20
- import { verseSessionRoots } from '../types.js';
39
+ import { canonicalModelId, claudeAutocompactFlag, hasExpansiveMode, } from '../context-math.js';
40
+ import { legacyModelOptionFallback } from '../model-windows.js';
41
+ import { VERSE_DEFAULT_CONTEXT_WINDOWS, verseSessionRoots, } from '../types.js';
21
42
  function isObject(value) {
22
43
  return typeof value === 'object' && value !== null && !Array.isArray(value);
23
44
  }
@@ -27,6 +48,12 @@ function num(value) {
27
48
  function str(value) {
28
49
  return typeof value === 'string' ? value : '';
29
50
  }
51
+ function positiveInt(value) {
52
+ return typeof value === 'number' && Number.isFinite(value) && value > 0 ? Math.floor(value) : null;
53
+ }
54
+ function nonNegativeInt(value) {
55
+ return typeof value === 'number' && Number.isFinite(value) && value >= 0 ? Math.floor(value) : null;
56
+ }
30
57
  /** Parse one JSONL line into an object, or null for anything that is not one. */
31
58
  export function parseJsonObjectLine(line) {
32
59
  const trimmed = line.trim();
@@ -55,15 +82,112 @@ function readAnthropicUsage(usage) {
55
82
  cacheCreation: num(usage['cache_creation_input_tokens']),
56
83
  };
57
84
  }
58
- function toVerseUsage(totals, last) {
85
+ const ZERO_USAGE = { input: 0, output: 0, cacheRead: 0, cacheCreation: 0 };
86
+ function addUsage(a, b) {
87
+ return {
88
+ input: a.input + b.input,
89
+ output: a.output + b.output,
90
+ cacheRead: a.cacheRead + b.cacheRead,
91
+ cacheCreation: a.cacheCreation + b.cacheCreation,
92
+ };
93
+ }
94
+ /**
95
+ * Two readings of the SAME API call. Every bucket is either fixed for the call
96
+ * (the prompt side, known at `message_start`) or cumulative within it (output,
97
+ * which `message_delta` and each `assistant` envelope re-report as it grows),
98
+ * so the per-bucket maximum is the call's final figure however many times and
99
+ * in whatever order it was reported.
100
+ */
101
+ function maxUsage(a, b) {
102
+ return {
103
+ input: Math.max(a.input, b.input),
104
+ output: Math.max(a.output, b.output),
105
+ cacheRead: Math.max(a.cacheRead, b.cacheRead),
106
+ cacheCreation: Math.max(a.cacheCreation, b.cacheCreation),
107
+ };
108
+ }
109
+ function isZeroUsage(u) {
110
+ return u.input === 0 && u.output === 0 && u.cacheRead === 0 && u.cacheCreation === 0;
111
+ }
112
+ function toVerseUsage(totals, contextTokens, contextWindow) {
59
113
  return {
60
114
  inputTokens: totals.input,
61
115
  outputTokens: totals.output,
62
116
  cacheReadTokens: totals.cacheRead,
63
117
  cacheCreationTokens: totals.cacheCreation,
64
- // Live context occupancy = prompt size of the most recent API call.
65
- contextTokens: last.input + last.cacheRead + last.cacheCreation,
66
- contextWindow: null,
118
+ // Live context occupancy = prompt size of the most recent API call. NOT
119
+ // clamped to the window: a reading above it is information.
120
+ contextTokens,
121
+ // The CLI's own figure for THIS turn (result.modelUsage), else null. The
122
+ // engine decides whether it wins (it ignores it on engine=local, where
123
+ // Verse set the window itself through CLAUDE_CODE_MAX_CONTEXT_TOKENS).
124
+ contextWindow,
125
+ };
126
+ }
127
+ /** `claude-opus-4-8[1m]` → `claude-opus-4-8`: the suffix is a request for the 1M window, not a different model. */
128
+ function stripOneMillionSuffix(id) {
129
+ return id.replace(/\[1m\]$/i, '');
130
+ }
131
+ /**
132
+ * The context window the CLI reported for this turn's model in
133
+ * `result.modelUsage` — the same number it compacts against, computed at
134
+ * result time, so it already reflects a long-context credit clamp (1M → 200k)
135
+ * or grok's mid-session window upgrade.
136
+ *
137
+ * Which row: `modelUsage` is keyed per model and a turn can touch several
138
+ * (claude runs side calls on Haiku). So, in order:
139
+ * 1. the row whose key IS the turn's model (canonicalised — a record naming
140
+ * the retired alias `claude-opus-5.5` still finds `claude-opus-5-5`);
141
+ * 2. the same comparison with a `[1m]` suffix stripped from both sides;
142
+ * 3. the ONLY row carrying a numeric `contextWindow` — grok's key can differ
143
+ * from the CLI id (`grok-4.6-build` for `grok-4.6`) and grok puts the
144
+ * window on the current model's row alone.
145
+ * Anything else (several windowed rows, none matching) is ambiguous and yields
146
+ * null: an unknown window is shown as unknown, never guessed.
147
+ */
148
+ export function runtimeContextWindow(modelUsage, model) {
149
+ if (!isObject(modelUsage))
150
+ return null;
151
+ const rows = Object.entries(modelUsage).filter((entry) => isObject(entry[1]));
152
+ if (rows.length === 0)
153
+ return null;
154
+ if (model) {
155
+ const wanted = canonicalModelId(model);
156
+ const exact = rows.find(([key]) => canonicalModelId(key) === wanted);
157
+ const exactWindow = exact ? positiveInt(exact[1]['contextWindow']) : null;
158
+ if (exactWindow !== null)
159
+ return exactWindow;
160
+ const wantedBase = stripOneMillionSuffix(wanted);
161
+ const stripped = rows.find(([key]) => stripOneMillionSuffix(canonicalModelId(key)) === wantedBase);
162
+ const strippedWindow = stripped ? positiveInt(stripped[1]['contextWindow']) : null;
163
+ if (strippedWindow !== null)
164
+ return strippedWindow;
165
+ }
166
+ const windowed = rows.map(([, row]) => positiveInt(row['contextWindow'])).filter((w) => w !== null);
167
+ return windowed.length === 1 ? windowed[0] : null;
168
+ }
169
+ /**
170
+ * `system/compact_boundary` → a `compaction` event. Both CLIs emit it (claude
171
+ * as `compact_metadata:{trigger, pre_tokens, post_tokens?, duration_ms?}`;
172
+ * grok documents the same line and ships the same field names). The camelCase
173
+ * spelling Claude Code uses in its own transcripts is accepted too, so a
174
+ * transcript-shaped line is never silently lost. Counts the CLI omitted stay
175
+ * null — the UI then says "compacted" without inventing a size.
176
+ */
177
+ function compactionFrom(turnId, ev) {
178
+ const meta = isObject(ev['compact_metadata'])
179
+ ? ev['compact_metadata']
180
+ : isObject(ev['compactMetadata']) ? ev['compactMetadata'] : {};
181
+ const pick = (snake, camel) => nonNegativeInt(meta[snake]) ?? nonNegativeInt(meta[camel]);
182
+ return {
183
+ type: 'compaction',
184
+ turnId,
185
+ // Only `manual` is distinguishable from the default; a headless turn that
186
+ // compacted without being asked to did so automatically.
187
+ trigger: meta['trigger'] === 'manual' ? 'manual' : 'auto',
188
+ preTokens: pick('pre_tokens', 'preTokens'),
189
+ postTokens: pick('post_tokens', 'postTokens'),
190
+ durationMs: pick('duration_ms', 'durationMs'),
67
191
  };
68
192
  }
69
193
  /** Render a tool_result `content` field (string or content-block array) as text. */
@@ -91,12 +215,22 @@ function toolResultText(content) {
91
215
  * - bare wire events (`message_start`, `content_block_start`, `content_block_delta`,
92
216
  * `content_block_stop`, `message_delta`, `message_stop`) — grok's streaming-messages-json;
93
217
  * - the same events wrapped in claude's `{type:'stream_event', event:{...}}`;
94
- * - claude's whole-message envelopes `assistant` / `user` and the terminal `result`;
95
- * - `system` (`init` captures session_id).
218
+ * - claude's whole-message envelopes `assistant` / `user` and the terminal `result`
219
+ * (whose `modelUsage` carries the CLI's context window for the turn);
220
+ * - `system` (`init` captures session_id and model; `compact_boundary`
221
+ * becomes a `compaction` event).
96
222
  *
97
223
  * A text/tool_use/thinking block can arrive twice — once via the streamed
98
224
  * content_block_* events and again inside an `assistant` envelope — so block
99
225
  * events are deduplicated by content within a turn.
226
+ *
227
+ * Usage is deduplicated the same way, by API call. With partial messages on,
228
+ * ONE call is reported by `message_start` + `message_delta` AND by one
229
+ * `assistant` envelope per content block. Summing every report (as this parser
230
+ * once did) counted a two-block call three times; the turn total only looked
231
+ * right because the terminal `result` replaced it. Each call is keyed by its
232
+ * message id and its readings are merged, so the fallback total is right too
233
+ * when a process dies before `result`.
100
234
  */
101
235
  /**
102
236
  * Shared by the claude/local and grok adapters — both speak the Anthropic wire
@@ -106,29 +240,54 @@ function toolResultText(content) {
106
240
  */
107
241
  export function createAnthropicStreamParser(turnId, engineLabel = 'claude') {
108
242
  let nativeId = null;
243
+ /** The model the CLI says it runs (`system/init.model`, else the latest assistant frame's). */
244
+ let initModel = null;
245
+ let frameModel = null;
109
246
  const open = new Map();
110
247
  const emittedText = new Set();
111
248
  const emittedThinking = new Set();
112
249
  const emittedToolUse = new Set();
113
250
  const emittedToolResult = new Set();
114
- let lastCallUsage = null;
115
- let turnTotals = { input: 0, output: 0, cacheRead: 0, cacheCreation: 0 };
251
+ /**
252
+ * One entry per API call, in the order the calls STARTED (Map keeps first
253
+ * insertion order on update), so the last entry is the latest call — whose
254
+ * prompt size is the live context occupancy.
255
+ */
256
+ const calls = new Map();
257
+ /** The call the wire stream is currently inside (set by `message_start`). */
258
+ let currentCall = null;
259
+ let syntheticCalls = 0;
116
260
  let resultTotals = null;
261
+ let runtimeWindow = null;
262
+ /**
263
+ * `post_tokens` of a compaction that happened AFTER the latest call. The
264
+ * latest call's prompt size predates it, so the usage event reports the
265
+ * CLI's post-compaction figure instead (a manual `/compact` turn makes no
266
+ * further call). Cleared as soon as another call starts.
267
+ */
268
+ let postCompactionTokens = null;
117
269
  let usageEmitted = false;
118
- // message_start carries input-side usage, message_delta the output side; they
119
- // describe the same API call, so merge them before rolling into the totals.
120
- let pendingCall = null;
121
- function commitPendingCall() {
122
- if (!pendingCall)
123
- return;
124
- lastCallUsage = pendingCall;
125
- turnTotals = {
126
- input: turnTotals.input + pendingCall.input,
127
- output: turnTotals.output + pendingCall.output,
128
- cacheRead: turnTotals.cacheRead + pendingCall.cacheRead,
129
- cacheCreation: turnTotals.cacheCreation + pendingCall.cacheCreation,
130
- };
131
- pendingCall = null;
270
+ function syntheticKey(kind) {
271
+ syntheticCalls += 1;
272
+ return `${kind}#${syntheticCalls}`;
273
+ }
274
+ function recordCall(key, usage) {
275
+ const previous = calls.get(key);
276
+ if (!previous)
277
+ postCompactionTokens = null;
278
+ calls.set(key, previous ? maxUsage(previous, usage) : usage);
279
+ }
280
+ function lastCall() {
281
+ let last = null;
282
+ for (const usage of calls.values())
283
+ last = usage;
284
+ return last;
285
+ }
286
+ function callTotals() {
287
+ let totals = ZERO_USAGE;
288
+ for (const usage of calls.values())
289
+ totals = addUsage(totals, usage);
290
+ return totals;
132
291
  }
133
292
  function emitText(out, text) {
134
293
  if (!text || emittedText.has(text))
@@ -193,23 +352,30 @@ export function createAnthropicStreamParser(turnId, engineLabel = 'claude') {
193
352
  function emitUsage(out) {
194
353
  if (usageEmitted)
195
354
  return;
196
- commitPendingCall();
197
- const totals = resultTotals ?? turnTotals;
198
- const last = lastCallUsage ?? resultTotals;
199
- if (!last)
355
+ const summed = callTotals();
356
+ // `result.usage` is the CLI's own turn total and wins — except when it is
357
+ // all zeros while calls were observed: grok documents an all-zero result
358
+ // usage as "unknown", not "free", so the observed calls are the better truth.
359
+ const totals = resultTotals && !(isZeroUsage(resultTotals) && !isZeroUsage(summed)) ? resultTotals : summed;
360
+ const last = lastCall() ?? resultTotals;
361
+ if (!last && postCompactionTokens === null)
200
362
  return;
201
363
  usageEmitted = true;
202
- out.push({ type: 'usage', turnId, usage: toVerseUsage(totals, last) });
364
+ const contextTokens = postCompactionTokens
365
+ ?? (last ? last.input + last.cacheRead + last.cacheCreation : 0);
366
+ out.push({ type: 'usage', turnId, usage: toVerseUsage(totals, contextTokens, runtimeWindow) });
203
367
  }
204
368
  function handleWireEvent(out, ev) {
205
369
  const type = str(ev['type']);
206
370
  switch (type) {
207
371
  case 'message_start': {
208
- commitPendingCall();
209
372
  const message = isObject(ev['message']) ? ev['message'] : null;
373
+ currentCall = str(message?.['id']) || syntheticKey('wire');
374
+ if (typeof message?.['model'] === 'string' && message['model'])
375
+ frameModel = message['model'];
210
376
  const usage = readAnthropicUsage(message?.['usage']);
211
377
  if (usage)
212
- pendingCall = usage;
378
+ recordCall(currentCall, usage);
213
379
  open.clear();
214
380
  return;
215
381
  }
@@ -258,24 +424,17 @@ export function createAnthropicStreamParser(turnId, engineLabel = 'claude') {
258
424
  case 'message_delta': {
259
425
  const usage = readAnthropicUsage(ev['usage']);
260
426
  if (usage) {
261
- if (pendingCall) {
262
- pendingCall = {
263
- input: usage.input || pendingCall.input,
264
- output: usage.output || pendingCall.output,
265
- cacheRead: usage.cacheRead || pendingCall.cacheRead,
266
- cacheCreation: usage.cacheCreation || pendingCall.cacheCreation,
267
- };
268
- }
269
- else {
270
- pendingCall = usage;
271
- }
427
+ if (!currentCall)
428
+ currentCall = syntheticKey('wire');
429
+ recordCall(currentCall, usage);
272
430
  }
273
431
  return;
274
432
  }
275
433
  case 'message_stop': {
276
434
  for (const index of [...open.keys()])
277
435
  closeBlock(out, index);
278
- commitPendingCall();
436
+ // `currentCall` stays set: an id-less `assistant` envelope that follows
437
+ // describes this same call, not a new one.
279
438
  return;
280
439
  }
281
440
  default:
@@ -286,8 +445,18 @@ export function createAnthropicStreamParser(turnId, engineLabel = 'claude') {
286
445
  const type = str(ev['type']);
287
446
  switch (type) {
288
447
  case 'system': {
289
- if (ev['subtype'] === 'init' && typeof ev['session_id'] === 'string')
290
- nativeId = ev['session_id'];
448
+ const subtype = ev['subtype'];
449
+ if (subtype === 'init') {
450
+ if (typeof ev['session_id'] === 'string')
451
+ nativeId = ev['session_id'];
452
+ if (typeof ev['model'] === 'string' && ev['model'])
453
+ initModel = ev['model'];
454
+ }
455
+ else if (subtype === 'compact_boundary') {
456
+ const compaction = compactionFrom(turnId, ev);
457
+ postCompactionTokens = compaction.postTokens;
458
+ out.push(compaction);
459
+ }
291
460
  return true;
292
461
  }
293
462
  case 'stream_event': {
@@ -301,16 +470,12 @@ export function createAnthropicStreamParser(turnId, engineLabel = 'claude') {
301
470
  if (!message)
302
471
  return true;
303
472
  if (type === 'assistant') {
473
+ if (typeof message['model'] === 'string' && message['model'])
474
+ frameModel = message['model'];
304
475
  const usage = readAnthropicUsage(message['usage']);
305
476
  if (usage) {
306
- pendingCall = null;
307
- lastCallUsage = usage;
308
- turnTotals = {
309
- input: turnTotals.input + usage.input,
310
- output: turnTotals.output + usage.output,
311
- cacheRead: turnTotals.cacheRead + usage.cacheRead,
312
- cacheCreation: turnTotals.cacheCreation + usage.cacheCreation,
313
- };
477
+ const key = str(message['id']) || currentCall || syntheticKey('envelope');
478
+ recordCall(key, usage);
314
479
  }
315
480
  }
316
481
  const content = message['content'];
@@ -331,6 +496,7 @@ export function createAnthropicStreamParser(turnId, engineLabel = 'claude') {
331
496
  const usage = readAnthropicUsage(ev['usage']);
332
497
  if (usage)
333
498
  resultTotals = usage;
499
+ runtimeWindow = runtimeContextWindow(ev['modelUsage'], initModel ?? frameModel);
334
500
  const subtype = str(ev['subtype']);
335
501
  if (subtype && subtype !== 'success') {
336
502
  const detail = str(ev['error']) || str(ev['result']) || subtype;
@@ -385,14 +551,94 @@ export function createAnthropicStreamParser(turnId, engineLabel = 'claude') {
385
551
  export function anthropicEnvBaseUrl(baseUrl) {
386
552
  return baseUrl.replace(/\/+$/, '').replace(/\/v1$/, '');
387
553
  }
554
+ // ---------------------------------------------------------------------------
555
+ // Launch helpers (pure: they read only the session and its launch snapshot)
556
+ // ---------------------------------------------------------------------------
557
+ /** The seat option for the session's model — exact id first, then canonical id (alias ↔ real id). */
558
+ function seatModelOption(session, launch) {
559
+ const models = Array.isArray(launch.seat?.models) ? launch.seat.models : [];
560
+ const exact = models.find((m) => m.id === session.model);
561
+ if (exact)
562
+ return exact;
563
+ const wanted = canonicalModelId(session.model);
564
+ return models.find((m) => canonicalModelId(m.id) === wanted) ?? null;
565
+ }
566
+ /**
567
+ * The option whose budgets decide a CLAUDE seat's `--autocompact` flag.
568
+ *
569
+ * Normally the launch snapshot's own option. A snapshot written before 3.9
570
+ * carries only the old flat 200k window and no budgets, so its model is looked
571
+ * up in the verified per-model table (model-windows.ts) instead — through
572
+ * `legacyModelOptionFallback`, the SAME helper the engine's
573
+ * `effectiveModelOption` calls, so the compaction point the engine records is
574
+ * exactly the one this flag tells the CLI. A model neither knows keeps its
575
+ * snapshot option (and so, lacking an expansive budget, gets no flag: the
576
+ * CLI's own default applies).
577
+ *
578
+ * The MODE is the session's. A pre-3.9 record has none on disk; the engine
579
+ * materialises one (`expansive` for a 1M model — what 3.8 ran) before it ever
580
+ * builds a launch, so this adapter never has to guess what "absent" meant.
581
+ */
582
+ function claudeBudgetOption(session, launch) {
583
+ return legacyModelOptionFallback('claude', session.model, seatModelOption(session, launch));
584
+ }
585
+ /**
586
+ * `--autocompact` for a claude seat, or nothing.
587
+ *
588
+ * Only 1M-native models (the ones with a real expansive budget) are steered:
589
+ * standard mode caps compaction at `claudeAutocompactFlag` (400k), expansive
590
+ * passes `auto` — explicitly, so a stray `autoCompactWindow` in the seat's own
591
+ * settings cannot silently shrink a budget the operator chose. A 200k model
592
+ * gets no flag at all; the CLI's `auto` already is its budget.
593
+ */
594
+ function autocompactArgs(session, launch) {
595
+ if (session.engine !== 'claude')
596
+ return [];
597
+ const option = claudeBudgetOption(session, launch);
598
+ if (!hasExpansiveMode(option))
599
+ return [];
600
+ const flag = claudeAutocompactFlag(option, session.contextMode ?? 'standard');
601
+ return ['--autocompact', flag === null ? 'auto' : String(flag)];
602
+ }
603
+ /**
604
+ * The window a LOCAL seat's CLI is told it has. The session's own window first
605
+ * (set from the seat's resolved num_ctx at creation, and never replaced by a
606
+ * runtime reading on engine=local), then the launch snapshot's option, then the
607
+ * seat's, then the named default the local seats pin.
608
+ */
609
+ function localContextWindow(session, launch) {
610
+ return positiveInt(session.usage?.contextWindow)
611
+ ?? positiveInt(seatModelOption(session, launch)?.contextWindow)
612
+ ?? positiveInt(launch.seat?.contextWindow)
613
+ ?? VERSE_DEFAULT_CONTEXT_WINDOWS['local'];
614
+ }
615
+ /** A launch record's memory snapshot, if it is well-formed; anything else is treated as "memory off". */
616
+ function launchMemory(launch) {
617
+ const memory = launch.memory;
618
+ if (!isObject(memory))
619
+ return null;
620
+ const dir = memory['dir'];
621
+ const block = memory['block'];
622
+ if (typeof dir !== 'string' || !dir.startsWith('/') || typeof block !== 'string' || block.trim().length === 0)
623
+ return null;
624
+ return { dir, block, writable: memory['writable'] === true };
625
+ }
388
626
  function buildClaudeLaunch(session, text, launch) {
389
627
  if (!session.nativeSessionId) {
390
628
  throw new Error('claude session is missing its native session id');
391
629
  }
392
- const prefix = session.engine === 'local' || !launch.launcher ? ['claude'] : [...launch.launcher];
630
+ const isLocal = session.engine === 'local';
631
+ const prefix = isLocal || !launch.launcher ? ['claude'] : [...launch.launcher];
393
632
  // Workspace roots beyond the primary. `verseSessionRoots` puts the primary
394
633
  // first and it is already the cwd, so only the tail needs a flag.
395
634
  const extraRoots = verseSessionRoots(session).slice(1);
635
+ const memory = launchMemory(launch);
636
+ // The memory directory is granted only when the snapshot says this seat may
637
+ // WRITE it: `--add-dir` has no read-only form, and a read-only seat already
638
+ // has the file's contents in the appended block.
639
+ const addDirs = [...extraRoots];
640
+ if (memory?.writable && !addDirs.includes(memory.dir) && memory.dir !== session.projectPath)
641
+ addDirs.push(memory.dir);
396
642
  // `-p` is boolean (`--print`); the prompt is the positional `[prompt]`. It
397
643
  // goes LAST, behind the end-of-options marker, so a message that starts
398
644
  // with `-` (a bullet list, or a literal `--dangerously-skip-permissions`)
@@ -403,10 +649,17 @@ function buildClaudeLaunch(session, text, launch) {
403
649
  '--output-format', 'stream-json',
404
650
  '--verbose',
405
651
  '--include-partial-messages',
406
- '--model', session.model,
652
+ // Always the CANONICAL id: `claude-opus-5.5` (an id Verse once shipped) is
653
+ // fuzzy-matched by the CLI to Opus 5, so a stored alias must never reach it.
654
+ '--model', canonicalModelId(session.model),
407
655
  '--permission-mode', 'acceptEdits',
408
656
  '--strict-mcp-config',
409
657
  '--mcp-config', '{"mcpServers":{}}',
658
+ ...autocompactArgs(session, launch),
659
+ // Local only: move the per-launch sections (cwd, env, git status) out of
660
+ // the system prompt so the local runner's prefix cache survives between
661
+ // turns. Claude seats keep the CLI default.
662
+ ...(isLocal ? ['--exclude-dynamic-system-prompt-sections'] : []),
410
663
  // VERIFIED against `claude --help` on 2.1.280:
411
664
  // `--add-dir <directories...> Additional directories to allow tool access to`
412
665
  // It is variadic, so it is spelled ONE DIRECTORY PER FLAG rather than
@@ -418,17 +671,25 @@ function buildClaudeLaunch(session, text, launch) {
418
671
  // option"), and `claude --add-dir /tmp --add-dir /private/var/tmp -p
419
672
  // --session-id NOTAUUID -- hi` got PAST option parsing to the downstream
420
673
  // "Invalid session ID" check, so both flags parsed.
421
- ...extraRoots.flatMap((dir) => ['--add-dir', dir]),
674
+ ...addDirs.flatMap((dir) => ['--add-dir', dir]),
675
+ // The `=` spelling binds the block to the flag whatever its first
676
+ // character is — commander never re-reads it as an option.
677
+ ...(memory ? [`--append-system-prompt=${memory.block}`] : []),
422
678
  ...(session.turnCount > 0 ? ['--resume', session.nativeSessionId] : ['--session-id', session.nativeSessionId]),
423
679
  '--', text,
424
680
  ];
425
- const env = session.engine === 'local'
681
+ const env = isLocal
426
682
  ? {
427
683
  // The launch record's dispatch address wins; `ollamaBaseUrl` is the
428
684
  // default lane and the fallback for records written before lanes existed.
429
685
  ANTHROPIC_BASE_URL: anthropicEnvBaseUrl(launch.anthropicBaseUrl ?? launch.ollamaBaseUrl),
430
686
  ANTHROPIC_AUTH_TOKEN: 'ollama',
431
687
  CLAUDE_CODE_DISABLE_NONESSENTIAL_TRAFFIC: '1',
688
+ // An Ollama tag is not a Claude id, so the CLI honours this as the
689
+ // model's window: auto-compaction and `result.modelUsage.contextWindow`
690
+ // then match what the runner actually serves. Not credential-shaped
691
+ // (`_TOKENS`), so the engine's env filter passes it through.
692
+ CLAUDE_CODE_MAX_CONTEXT_TOKENS: String(localContextWindow(session, launch)),
432
693
  }
433
694
  : {};
434
695
  return { argv, cwd: session.projectPath, env, stdin: null };