docks-kit 0.17.2 → 0.18.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -6,26 +6,56 @@ against.
6
6
 
7
7
  ## Role map
8
8
 
9
- | Role | Model | Level | Index | Cost/task | TTFT | Coding Agent Index |
10
- |---|---|---|---:|---:|---:|---:|
11
- | `default` | `anthropic/claude-opus-5` | high | 48 | $3.61 | 16.96 s | 66 (Claude Code) |
12
- | `slow` | `anthropic/claude-opus-5` | xhigh | 50 | $4.88 | 28.65 s | 68 (Claude Code) |
13
- | `plan` | `anthropic/claude-opus-5` | xhigh | 50 | $4.88 | 28.65 s | 68 (Claude Code) |
14
- | `task` | `openai-codex/gpt-5.6-sol` | high | 42 | $0.81 | 11.26 s | 64 (Codex) |
15
- | `advisor` | `openai-codex/gpt-5.6-sol` | medium | 39 | $0.50 | 4.90 s | 62 (Codex) |
16
- | `designer` | `anthropic/claude-opus-5` | high | 48 | $3.61 | 16.96 s | 66 (Claude Code) |
17
- | `vision` | `anthropic/claude-opus-5` | medium | 45 | $2.19 | 3.79 s | 64 (Claude Code) |
18
- | `smol` / `commit` | `openai-codex/gpt-5.6-luna` | medium | 26 | n/a | 2.18 s | 42 (Codex) |
19
- | `tiny` | `openai-codex/gpt-5.6-luna` | low | 22 | n/a | 1.78 s | 25 (Codex) |
20
- | `fable` | `anthropic/claude-fable-5-1` | medium | 49 | $2.98 | 9.65 s | n/a |
21
- | `switch_fable` | `anthropic/claude-fable-5-1` | medium | 49 | $2.98 | 9.65 s | n/a |
22
- | `astra` | `openai-codex/gpt-6-astra` | xhigh | 53 | $2.31 | 161.65 s | n/a |
9
+ | Role | Model | Level | Index | Cost/task | TTFT |
10
+ |---|---|---|---:|---:|---:|
11
+ | `default` | `anthropic/claude-opus-5-5` | high | 54 | $1.82 | 12.49 s |
12
+ | `slow` | `anthropic/claude-opus-5-5` | xhigh | 56 | $3.46 | 165.20 s |
13
+ | `plan` | `anthropic/claude-opus-5-5` | xhigh | 56 | $3.46 | 165.20 s |
14
+ | `task` | `openai-codex/gpt-6-sol` | high | 43 | $0.37 | n/a |
15
+ | `advisor` | `openai-codex/gpt-6-sol` | medium | 40 | $0.25 | n/a |
16
+ | `designer` | `anthropic/claude-opus-5-5` | high | 54 | $1.82 | 12.49 s |
17
+ | `vision` | `anthropic/claude-opus-5-5` | medium | 51 | $1.34 | 22.17 s |
18
+ | `smol` / `commit` | `openai-codex/gpt-6-luna` | medium | 29 | $0.02 | n/a |
19
+ | `tiny` | `openai-codex/gpt-6-luna` | low | 21 | $0.0045 | n/a |
20
+ | `fable` | `anthropic/claude-fable-5-1` | medium | 49 | $2.98 | 8.61 s |
21
+ | `switch_fable` | `anthropic/claude-fable-5-1` | medium | 49 | $2.98 | 8.61 s |
22
+ | `astra` | `openai-codex/gpt-6-astra` | xhigh | 52 | $2.31 | 188.20 s |
23
+ | `web` | `web/firecrawl` | n/a | n/a | n/a | n/a |
23
24
 
24
25
  The table reports the measured Artificial Analysis figures for each assigned
25
26
  model and level. It states no motive that the config or omp's own
26
- documentation does not establish. AA lists no cost per task for Luna medium
27
- and low, and no Coding Agent Index for any Astra or Fable 5.1 level except
28
- max.
27
+ documentation does not establish. AA has not measured output speed or latency
28
+ for GPT-6 Sol or GPT-6 Luna at any level, so those rows carry `n/a` for TTFT.
29
+ AA measures no web search provider, so the `web` row carries no figures.
30
+
31
+ The role map carries no Coding Agent Index column. That index publishes one
32
+ entry per harness and model, and the Codex entries for GPT-6 Sol and GPT-6
33
+ Luna run at `max`, a level no role here uses. The snapshot section below
34
+ lists the entries.
35
+
36
+ ### GPT-6 Sol and GPT-6 Luna availability
37
+
38
+ The `openai-codex` selectors for GPT-6 Sol and GPT-6 Luna were deployed on
39
+ 2026-09-22, while the rollout of both models was still in progress. The owner
40
+ chose to deploy them before the rollout completed. Checks on that date, with
41
+ omp 18.2.9 and codex-cli 0.153.3:
42
+
43
+ - At 19:02 UTC, `codex exec -m gpt-6-sol` returned HTTP 400, `The 'gpt-6-sol'
44
+ model is not supported when using Codex with a ChatGPT account.`
45
+ `gpt-6-luna` returned the same error. The Codex model cache fetched at that
46
+ time listed neither model.
47
+ - At 19:14 UTC, the Codex model cache for the same account listed
48
+ `gpt-6-sol` and `gpt-6-luna`. A request on the new default could not be
49
+ tested, because the account had reached its Codex usage limit.
50
+ - The omp `openai-codex` catalog listed `gpt-6-astra` and the three `gpt-5.6`
51
+ models, but not `gpt-6-sol` or `gpt-6-luna`, both before and after
52
+ `omp models refresh` at 19:14 UTC. omp resolves a selector that is not in
53
+ the catalog by provider-scoped fuzzy match. An `omp -p --mode json` run on
54
+ `openai-codex/gpt-6-sol:low` recorded `gpt-5.6-sol` as the serving model,
55
+ and the Luna selector recorded `gpt-5.6-luna`. omp printed no warning.
56
+
57
+ Until the omp catalog lists both ids, the omp roles run GPT-5.6. To check, run
58
+ `omp models openai-codex`: the `gpt-6-sol` and `gpt-6-luna` rows must appear.
29
59
 
30
60
  What omp's settings catalog establishes about these roles:
31
61
 
@@ -42,145 +72,235 @@ and `security-reviewer` exist as OMP agents (omp's task tool lists `scout`,
42
72
  `reviewer`, `security-reviewer`, `task`, and `sonic`), so those two inherit
43
73
  whatever `task` resolves to. The `code-reviewer` and `plan-reviewer` entries
44
74
  are dormant until an OMP agent with that name exists.
45
- `retry.fallbackChains.task` keeps `anthropic/claude-opus-5:high` as a
75
+ `retry.fallbackChains.task` keeps `anthropic/claude-opus-5-5:high` as a
46
76
  cross-vendor fallback.
47
77
 
48
78
  `retry.fallbackChains.astra` holds `anthropic/claude-fable-5-1:medium`.
49
79
  `retry.fallbackChains.fable` holds `openai-codex/gpt-6-astra:xhigh`.
50
80
  Each deliberate cycle stop falls to the other vendor. Without these explicit
51
81
  chains, `retry.fallbackChains.default` would send either stop to
52
- `openai-codex/gpt-5.6-sol:high`.
82
+ `openai-codex/gpt-6-sol:high`.
53
83
  Chain entries are concrete selectors, not role aliases, so this pair cannot
54
84
  recurse. The hidden `switch_fable` chain stays empty.
55
85
 
86
+ `modelRoles.web` is `web/firecrawl`, and `retry.fallbackChains.web` lists the
87
+ explicit 20-entry provider order that follows it. The two keys replace the
88
+ retired `providers.webSearchOrder` key, which omp no longer carries in its
89
+ settings schema. omp still accepts that key in a deployed file, expands it in
90
+ memory into the same two keys, and then drops it, but it never writes the
91
+ expansion back to disk. The kit therefore declares both keys itself.
92
+
93
+ The chain starts from the expansion omp produces, read back with
94
+ `omp config get retry.fallbackChains`. An explicit chain replaces omp's
95
+ built-in web order wholesale, so a shortened list drops providers instead of
96
+ reordering them. The kit makes two deliberate edits to omp's order:
97
+
98
+ - The Codex Luna entry follows the kit's Luna generation and names
99
+ `openai-codex/gpt-6-luna`.
100
+ - The owner removed every entry that named an older model:
101
+ `google/gemini-2.5-flash`, `google-antigravity/gemini-2.5-flash`,
102
+ `anthropic/claude-haiku-4-5`, `openai-codex/gpt-5.6`,
103
+ `openai-codex/gpt-5.5`, `xai/grok-4.5`, and `xai-oauth/grok-4.5`. omp never
104
+ tries those search backends now.
105
+
106
+ The order keeps Firecrawl, Exa, Perplexity, and Codex first. The remaining
107
+ `web/*` entries are omp's own ordering of the providers behind them.
108
+
56
109
  ## Artificial Analysis snapshot
57
110
 
58
- Source: `https://artificialanalysis.ai`, read on 2026-09-08. Every score below
59
- comes from one snapshot: Intelligence Index v4.3 and Coding Agent Index v1.4,
60
- taken from each family's release page, its per-level model pages, and the
61
- harness comparison pages. The v4.1.1-era figures in AA's Astra launch article
62
- are excluded, because index composition changed in v4.2 and again in v4.3, so
63
- mixing them would invalidate every ratio here. The head-to-head rows come from
64
- the direct `gpt-6-astra-low-vs-gpt-5-6-sol-high` comparison page, not from a
65
- comparison against another Sol level.
111
+ Source: `https://artificialanalysis.ai`, read on 2026-09-22. Every figure below
112
+ comes from that one capture at Intelligence Index v4.3.2 and Coding Agent Index
113
+ v1.5. Each per-level row was read from the metric table of the comparison page
114
+ `/models/comparisons/<level-slug>-vs-gpt-5-6-sol-high`. The page title named
115
+ the requested model and level, and the page printed index v4.3.2. AA serves
116
+ the `max` level under the bare model slug. Index composition changed in v4.2,
117
+ again in v4.3, and again in v4.3.2, so figures from an earlier capture cannot
118
+ be mixed with these.
119
+
120
+ Coding Agent Index v1.5 (`/agents/coding-agents`) lists 12 harness and model
121
+ entries. Most carry `max`. Grok Build with Grok 4.7 runs at `xhigh`, and
122
+ Antigravity SDK with Gemini 3.8 Flash runs at `high`. Opencode with GLM-5.3
123
+ and Kimi Code CLI with Kimi K3 name no level. The entries that involve a model
124
+ family in this topic:
125
+
126
+ | Harness and model | Coding Agent Index |
127
+ |---|---:|
128
+ | Claude Code - Fable 5.1 (max, with fallback) | 62.2 |
129
+ | Codex - GPT-6 Astra (max) | 61.6 |
130
+ | Claude Code - Opus 5 (max) | 59.7 |
131
+ | Codex - GPT-6 Sol (max) | 56.7 |
132
+ | Codex - GPT-6 Luna (max) | 41.1 |
133
+
134
+ AA lists no Opus 5.5 entry. The page stores each score as a fraction, such as
135
+ `0.6222`, and this table shows it multiplied by 100.
66
136
 
67
137
  Column meanings:
68
138
 
69
- - **Intelligence Index** - AA's weighted aggregate across its evaluation set.
70
- Comparable only inside one index version.
71
- - **Coding Agent Index** - agentic coding score inside a named harness. AA
72
- publishes it per harness, and for most effort levels it publishes nothing.
73
- - **Cost per Index task** - weighted average USD to run one index task,
74
- including input, cache, reasoning, and answer tokens.
75
- - **Index output tokens** - total output tokens the model spends to complete the
139
+ - **Index** - AA Intelligence Index, a weighted aggregate across its
140
+ evaluation set. Comparable only inside one index version.
141
+ - **Cost/task** - weighted average USD to run one index task, including
142
+ input, cache, reasoning, and answer tokens.
143
+ - **Tokens/task** - answer plus reasoning tokens for one index task.
144
+ - **Index tokens** - total output tokens the model spends to complete the
76
145
  whole index run. This is the token-efficiency signal.
146
+ - **Speed** - output tokens per second.
77
147
  - **TTFT** - seconds to the first answer token, so reasoning time counts.
148
+ - **TB 4.0** - Terminal-Bench 4.0, one of the ten index evaluations.
149
+
150
+ ### Claude Opus 5.5 (Anthropic) - `anthropic/claude-opus-5-5`
151
+
152
+ | Level | Index | Cost/task | Tokens/task | Index tokens | Speed t/s | TTFT s | TB 4.0 |
153
+ |---|---:|---:|---:|---:|---:|---:|---:|
154
+ | max | 58 | $5.98 | 119k | 260M | n/a | n/a | 60% |
155
+ | xhigh | 56 | $3.46 | 66k | 100M | 72 | 165.20 | 60% |
156
+ | high | 54 | $1.82 | 36k | 53M | 91 | 12.49 | 57% |
157
+ | medium | 51 | $1.34 | 26k | 38M | 76 | 22.17 | 53% |
158
+ | low | 42 | $0.55 | 10k | 20M | 94 | 4.79 | 31% |
159
+
160
+ Price: $4.00 in, $20.00 out, $0.20 cache hit per 1M. The Anthropic platform
161
+ documentation read the same day lists the same input, output, and cache-read
162
+ prices. It adds a $5.00 five-minute cache write and an $8.00 one-hour cache
163
+ write, which the AA comparison table does not show. Context 1M, maximum output
164
+ 128K. Adaptive thinking is always on, and the Claude API default effort is
165
+ `medium`. All AA levels run with fallback. Max is the highest Intelligence
166
+ Index in this topic at this capture. AA has not measured speed or latency for
167
+ max. It measures a longer TTFT for medium than for high.
168
+
169
+ `SoT/.omp/models.yml` carries the Anthropic limits and prices as an `anthropic`
170
+ `modelOverrides` block, because the shared catalog still serves this id as a
171
+ stub with null limits and zero cost. Remove that block once the catalog
172
+ publishes the row.
78
173
 
79
- ### GPT-6 Astra (OpenAI) - `openai-codex/gpt-6-astra`
174
+ ### Claude Fable 5.1 (Anthropic) - `anthropic/claude-fable-5-1`
80
175
 
81
- | Level | Intelligence Index | Coding Agent Index | Cost per Index task | Index output tokens | Output speed t/s | TTFT s |
82
- |---|---:|---:|---:|---:|---:|---:|
83
- | max | 53 | 67 (Codex) | $3.26 | 60M | 59 | 322.48 |
84
- | xhigh | 53 | n/a | $2.31 | 38M | 57 | 161.65 |
85
- | high | 51 | n/a | $1.72 | 26M | 55 | 45.63 |
86
- | medium | 50 | n/a | $1.54 | 19M | 53 | 5.42 |
87
- | low | 46 | n/a | $0.82 | 10M | 53 | 2.60 |
88
- | non-reasoning | 45 | n/a | $1.71 | 12M | n/a | n/a |
176
+ | Level | Index | Cost/task | Tokens/task | Index tokens | Speed t/s | TTFT s | TB 4.0 |
177
+ |---|---:|---:|---:|---:|---:|---:|---:|
178
+ | max | 53 | $7.63 | 78k | 188M | 66 | 311.51 | 52% |
179
+ | xhigh | 53 | $5.98 | 61k | 121M | 61 | 164.82 | 55% |
180
+ | high | 51 | $3.91 | 38k | 62M | 55 | 26.22 | 52% |
181
+ | medium | 49 | $2.98 | 28k | 44M | 55 | 8.61 | 45% |
182
+ | low | 47 | $2.37 | 22k | 33M | 54 | 7.28 | 40% |
89
183
 
90
- Price: $10.00 in, $50.00 out, $1.00 cache read, $12.50 cache write per 1M.
91
- Context 1M. Knowledge cutoff 2026-04-30.
92
- AA lists non-reasoning above low on cost per task.
184
+ Price: $10.00 in, $50.00 out, $0.25 cache hit per 1M. Context 1M. All levels
185
+ run with fallback.
93
186
 
94
- ### Claude Fable 5.1 (Anthropic) - `anthropic/claude-fable-5-1`
187
+ ### GPT-6 Astra (OpenAI) - `openai-codex/gpt-6-astra`
95
188
 
96
- | Level | Intelligence Index | Coding Agent Index | Cost per Index task | Index output tokens | Output speed t/s | TTFT s |
97
- |---|---:|---:|---:|---:|---:|---:|
98
- | max | 54 (estimated) | 70 (Claude Code) | n/a | n/a | 70 | 277.47 |
99
- | xhigh | 53 | n/a | $5.98 | n/a | 60 | 124.87 |
100
- | high | 51 | n/a | $3.91 | n/a | 57 | 23.70 |
101
- | medium | 49 | n/a | $2.98 | n/a | 56 | 9.65 |
102
- | low | 47 | n/a | $2.37 | n/a | 53 | 6.55 |
103
-
104
- Price: $10.00 in, $50.00 out, $0.25 cache read per 1M; no published cache-write
105
- price. Context 1M. All levels run with fallback.
106
- AA publishes per-task output tokens instead of index totals here: low 22k,
107
- medium 28k, high 38k, xhigh 61k. AA marks the max index score as estimated.
108
-
109
- ### Claude Opus 5 (Anthropic) - `anthropic/claude-opus-5`
110
-
111
- | Level | Intelligence Index | Coding Agent Index | Cost per Index task | Index output tokens | Output speed t/s | TTFT s |
112
- |---|---:|---:|---:|---:|---:|---:|
113
- | max | 51 | 67 (Claude Code) | $5.86 | 140M | 54.3 | 69.92 |
114
- | xhigh | 50 | 68 (Claude Code) | $4.88 | 110M | 53.0 | 28.65 |
115
- | high | 48 | 66 (Claude Code) | $3.61 | 81M | 54.0 | 16.96 |
116
- | medium | 45 | 64 (Claude Code) | $2.19 | 49M | 53.6 | 3.79 |
117
- | low | 40 | 59 (Claude Code) | $1.10 | 26M | 53.2 | 2.32 |
118
-
119
- Price: $5.00 in, $25.00 out, $0.50 cache read, $6.25 cache write per 1M, with a
120
- 5-minute cache TTL. Context 1M.
121
- AA reports that its Opus 5 index run fell back to Opus 4.8 for part of the set.
122
-
123
- ### GPT-5.6 Sol (OpenAI) - `openai-codex/gpt-5.6-sol`
124
-
125
- | Level | Intelligence Index | Coding Agent Index | Cost per Index task | Index output tokens | Output speed t/s | TTFT s |
126
- |---|---:|---:|---:|---:|---:|---:|
127
- | max | 47 | 65 (Codex) | $1.99 | 90M | 69.8 | 132.10 |
128
- | xhigh | 44 | 63 (Codex) | $1.18 | 51M | 64.8 | 50.59 |
129
- | high | 42 | 64 (Codex) | $0.81 | 34M | 67.8 | 11.26 |
130
- | medium | 39 | 62 (Codex) | $0.50 | 21M | 66.6 | 4.90 |
131
- | low | 34 | 55 (Codex) | $0.26 | 13M | 67.1 | 2.69 |
132
- | non-reasoning | 28 | 43 (Codex) | n/a | n/a | 65.6 | 1.13 |
133
-
134
- Price: $4.00 in, $20.00 out per 1M, with a 90% cache-read discount and no
135
- published numeric cache price. Context 1M.
136
-
137
- ### GPT-5.6 Luna (OpenAI) - `openai-codex/gpt-5.6-luna`
138
-
139
- | Level | Intelligence Index | Coding Agent Index | Cost per Index task | Index output tokens | Output speed t/s | TTFT s |
140
- |---|---:|---:|---:|---:|---:|---:|
141
- | max | 38 | 57 (Codex) | $0.18 | n/a | 121 | 168.22 |
142
- | xhigh | 35 | 53 (Codex) | $0.09 | n/a | 113 | 60.22 |
143
- | high | 33 | 52 (Codex) | n/a | n/a | 120 | 9.62 |
144
- | medium | 26 | 42 (Codex) | n/a | n/a | 110 | 2.18 |
145
- | low | 22 | 25 (Codex) | n/a | n/a | 119 | 1.78 |
146
- | non-reasoning | 17 | 19 (Codex) | n/a | n/a | 120 | 0.76 |
147
-
148
- Price: $0.20 in, $1.20 out, $0.02 cache read per 1M; no published cache-write
149
- price. Context 1M. AA lists cost per task only for max and xhigh.
150
-
151
- ## Why `task` returns to Sol high
152
-
153
- `task` runs `openai-codex/gpt-5.6-sol:high`. The owner uses Astra only for
154
- main orchestration, so Astra now has a dedicated `astra` cycle stop at `xhigh`.
155
- The comparison below records the retired Astra-low choice beside the current
156
- Sol-high choice and the other measured alternatives.
157
-
158
- | Metric | Astra low, retired task | Sol high, current task | Sol max | Opus 5 high |
159
- |---|---:|---:|---:|---:|
160
- | Intelligence Index | 46 | 42 | 47 | 48 |
161
- | Cost per Index task | $0.82 | $0.81 | $1.99 | $3.61 |
162
- | Output tokens per task | 4k | 13k | 29k | 46k |
163
- | Index output tokens | 10M | 34M | 90M | 81M |
164
- | Answer TTFT | 2.60 s | 11.26 s | 132.10 s | 16.96 s |
165
- | End-to-end latency | 11.99 s | 18.64 s | 139.27 s | 26.22 s |
166
- | Time per task | 84.45 s | 195.67 s | 411.54 s | 525.57 s |
167
- | Terminal-Bench v4.0 | 42% | 21% | 40% | 46% |
168
- | AA-Briefcase | 1253 | 1361 | 1475 | 1557 |
169
- | AA-Omniscience | 41 | 20 | 22 | 34 |
170
-
171
- Astra low beat Sol high on intelligence, latency, and token use at effectively
172
- equal cost per task. It cost $0.82 against $0.81 and used 4k output tokens
173
- per task against 13k. Its 2.60 s TTFT beat Sol high's 11.26 s.
174
- That was a latency and token-budget win, not a cost saving.
175
- The owner reversed that trade on purpose to reserve Astra for interactive
176
- orchestration.
177
-
178
- Sol high scores 64 in the Codex Coding Agent Index and 1361 on AA-Briefcase.
179
- Astra low scores 1253 on AA-Briefcase. AA publishes no Astra Coding Agent
180
- Index except max, which scores 67 in Codex.
181
- The new Astra xhigh cycle stop scores 53 on the Intelligence Index, costs
182
- $2.31 per index task, and has 161.65 s TTFT.
183
- It has no published Coding Agent Index.
189
+ | Level | Index | Cost/task | Tokens/task | Index tokens | Speed t/s | TTFT s | TB 4.0 |
190
+ |---|---:|---:|---:|---:|---:|---:|---:|
191
+ | max | 53 | $3.26 | 27k | 60M | 61 | 322.65 | 59% |
192
+ | xhigh | 52 | $2.31 | 17k | 38M | 55 | 188.20 | 60% |
193
+ | high | 51 | $1.73 | 12k | 26M | 50 | 79.00 | 54% |
194
+ | medium | 50 | $1.54 | 10k | 19M | 48 | 6.19 | 49% |
195
+ | low | 46 | $0.82 | 4k | 10M | 51 | 2.76 | 42% |
196
+
197
+ Price: $10.00 in, $50.00 out, $1.00 cache hit per 1M. Context 1M. Knowledge
198
+ cutoff 2026-04-30. AA publishes no non-reasoning Astra row.
199
+
200
+ ### GPT-6 Sol (OpenAI) - `openai-codex/gpt-6-sol`
201
+
202
+ | Level | Index | Cost/task | Tokens/task | Index tokens | Speed t/s | TTFT s | TB 4.0 |
203
+ |---|---:|---:|---:|---:|---:|---:|---:|
204
+ | max | 48 | $1.06 | 31k | 77M | n/a | n/a | 44% |
205
+ | xhigh | 44 | $0.53 | 16k | 40M | n/a | n/a | 30% |
206
+ | high | 43 | $0.37 | 10k | 25M | n/a | n/a | 26% |
207
+ | medium | 40 | $0.25 | 6k | 16M | n/a | n/a | 19% |
208
+ | low | 34 | $0.13 | 3k | 9M | n/a | n/a | 9% |
209
+ | non-reasoning | 28 | $0.33 | 5k | 8M | n/a | n/a | 13% |
210
+
211
+ Price: $2.00 in, $10.00 out, $0.20 cache hit per 1M. OpenAI's model page lists
212
+ a $2.50 cache write and a 1,050,000-token context with 128,000 maximum output
213
+ tokens. It bills a prompt above 272K input tokens at 2x input and cache rates
214
+ and 1.5x output for the full request. Knowledge cutoff 2026-04-20. The API
215
+ effort ladder is `none, low, medium, high, xhigh, max`, with `medium` as the
216
+ default. AA has not measured speed or latency for any level.
217
+
218
+ ### GPT-6 Luna (OpenAI) - `openai-codex/gpt-6-luna`
219
+
220
+ | Level | Index | Cost/task | Tokens/task | Index tokens | Speed t/s | TTFT s | TB 4.0 |
221
+ |---|---:|---:|---:|---:|---:|---:|---:|
222
+ | max | 37 | $0.07 | 51k | 145M | n/a | n/a | 13% |
223
+ | xhigh | 34 | $0.04 | 27k | 68M | n/a | n/a | 8% |
224
+ | high | 32 | $0.03 | 20k | 47M | n/a | n/a | 5% |
225
+ | medium | 29 | $0.02 | 11k | 28M | n/a | n/a | 3% |
226
+ | low | 21 | $0.0045 | 2k | 8M | n/a | n/a | 0% |
227
+ | non-reasoning | 18 | $0.01 | 4k | 7M | n/a | n/a | 2% |
228
+
229
+ Price: $0.10 in, $0.50 out, $0.01 cache hit per 1M. OpenAI's model page lists
230
+ a $0.125 cache write, the same context, output, long-prompt billing, and
231
+ effort ladder as Sol, and a 2026-05-18 knowledge cutoff. AA has not measured
232
+ speed or latency for any level.
233
+
234
+ ### GPT-5.6 Sol (OpenAI) - previous generation
235
+
236
+ The role map no longer uses GPT-5.6 Sol. This table stays as the measured
237
+ baseline for the GPT-6 Sol switch.
238
+
239
+ | Level | Index | Cost/task | Tokens/task | Index tokens | Speed t/s | TTFT s | TB 4.0 |
240
+ |---|---:|---:|---:|---:|---:|---:|---:|
241
+ | max | 47 | $1.99 | 29k | 90M | 82 | 130.17 | 40% |
242
+ | xhigh | 44 | $1.18 | 20k | 51M | 72 | 35.55 | 25% |
243
+ | high | 42 | $0.81 | 13k | 34M | 68 | 17.49 | 21% |
244
+ | medium | 39 | $0.50 | 8k | 21M | 58 | 5.07 | 15% |
245
+ | low | 33 | $0.26 | 4k | 13M | 58 | 3.85 | 1% |
246
+ | non-reasoning | 28 (estimated) | n/a | n/a | n/a | 64 | 1.22 | n/a |
247
+
248
+ Price: $4.00 in, $20.00 out, $0.40 cache hit per 1M. Context 1M. AA marks the
249
+ non-reasoning index score as estimated.
250
+
251
+ ### GPT-5.6 Luna (OpenAI) - previous generation
252
+
253
+ The role map no longer uses GPT-5.6 Luna. This table stays as the measured
254
+ baseline for the GPT-6 Luna switch.
255
+
256
+ | Level | Index | Cost/task | Tokens/task | Index tokens | Speed t/s | TTFT s | TB 4.0 |
257
+ |---|---:|---:|---:|---:|---:|---:|---:|
258
+ | max | 37 | $0.18 | 41k | 154M | 145 | 122.15 | 12% |
259
+ | xhigh | 35 | $0.09 | 24k | 85M | 143 | 40.24 | 4% |
260
+ | high | 32 | $0.04 | 14k | 50M | 132 | 15.15 | 3% |
261
+ | medium | 25 | $0.02 | 4k | 18M | 133 | 2.57 | 1% |
262
+ | low | 21 | $0.01 | 3k | 10M | 131 | 1.69 | 0% |
263
+ | non-reasoning | 16 | $0.01 | 2k | 5M | 140 | 0.81 | 1% |
264
+
265
+ Price: $0.20 in, $1.20 out, $0.02 cache hit per 1M. Context 1M.
266
+
267
+ ## Why `task` runs GPT-6 Sol high
268
+
269
+ `task` runs `openai-codex/gpt-6-sol:high`, the same level GPT-5.6 Sol ran
270
+ before it. The owner uses Astra only for main orchestration, so Astra has a
271
+ dedicated `astra` cycle stop at `xhigh`. The comparison below records the
272
+ retired Astra-low and GPT-5.6 Sol choices beside the current GPT-6 Sol choice.
273
+
274
+ | Metric | Astra low, retired | GPT-5.6 Sol high, previous | GPT-6 Sol high, current | GPT-6 Sol max | Opus 5.5 high, `default` |
275
+ |---|---:|---:|---:|---:|---:|
276
+ | Intelligence Index | 46 | 42 | 43 | 48 | 54 |
277
+ | Cost per Index task | $0.82 | $0.81 | $0.37 | $1.06 | $1.82 |
278
+ | Output tokens per task | 4k | 13k | 10k | 31k | 36k |
279
+ | Index output tokens | 10M | 34M | 25M | 77M | 53M |
280
+ | Answer TTFT | 2.76 s | 17.49 s | n/a | n/a | 12.49 s |
281
+ | End-to-end response time | 12.49 s | 24.87 s | n/a | n/a | 18.00 s |
282
+ | Time per index task | 87.88 s | 189.16 s | n/a | n/a | 244.35 s |
283
+ | Terminal-Bench 4.0 | 42% | 21% | 26% | 44% | 57% |
284
+ | AA-Briefcase v1.1 | 1261 | 1370 | 1289 | 1483 | 1705 |
285
+ | AA-Omniscience | 41 | 20 | 27 | 27 | 41 |
286
+
287
+ GPT-6 Sol high scores one index point above GPT-5.6 Sol high. It costs $0.37
288
+ against $0.81 per index task and uses 10k output tokens per task against 13k.
289
+ It also scores 26% against 21% on Terminal-Bench 4.0. AA-Briefcase is the one
290
+ metric where the older model leads, at 1370 against 1289. AA has not measured
291
+ GPT-6 Sol latency, so the latency comparison is open.
292
+
293
+ Astra low still beats GPT-6 Sol high on the index, at 46 against 43, and on
294
+ Terminal-Bench 4.0, at 42% against 26%. It costs $0.82 against $0.37 per index
295
+ task. The owner reserves Astra for interactive orchestration.
296
+
297
+ Opus 5.5 high scores 11 points above GPT-6 Sol high and costs $1.82 against
298
+ $0.37 per index task. It is the `default`, `designer`, and fallback model, not
299
+ the `task` model, so the subagent fan-out keeps the cheaper Sol.
300
+
301
+ The `astra` cycle stop runs xhigh: index 52, $2.31 per index task, and
302
+ 188.20 s TTFT. AA publishes a Coding Agent Index entry for Astra only at max,
303
+ paired with Codex, where it scores 61.6.
184
304
 
185
305
  ## How Astra is selected in practice
186
306
 
@@ -201,16 +321,17 @@ quick answer.
201
321
  not Sol or Astra. To move them, change `modelRoles.smol` or add a
202
322
  `task.agentModelOverrides` entry for the agent name.
203
323
  - The bundled `task` agent carries `model: "@task"` and
204
- `thinking-level: auto`. It resolves Sol, and `auto` classifies each prompt
324
+ `thinking-level: auto`. It resolves GPT-6 Sol, and `auto` classifies each prompt
205
325
  to choose a thinking level.
206
326
  - `task.enableEffort` is `true`, so a caller can pass `effort: lo`, `med`, or
207
327
  `hi`, which overrides `auto`.
208
- - `task.maxEffort` is `high`, so `scout` and `sonic` run Luna `medium` by
209
- default and Luna `high` with `effort: hi`.
210
- - The bundled `reviewer` and `security-reviewer` inherit `@task`, now Sol high.
328
+ - `task.maxEffort` is `max`, so `scout` and `sonic` run GPT-6 Luna `medium`
329
+ by default and GPT-6 Luna `max` with `effort: hi`.
330
+ - The bundled `reviewer` and `security-reviewer` inherit `@task`, now GPT-6
331
+ Sol high.
211
332
  - The `code-reviewer` and `plan-reviewer` override entries remain dormant.
212
- Both point to `@task`, now Sol high. omp's task tool rejects both names as
213
- unknown agents, so neither can spawn.
333
+ Both point to `@task`, now GPT-6 Sol high. omp's task tool rejects both
334
+ names as unknown agents, so neither can spawn.
214
335
 
215
336
  The runtime per-agent measurement from fresh `omp -p` runs on 2026-09-09
216
337
  predates this change. It applies to the retired Astra-low `task` configuration,
@@ -327,9 +448,13 @@ recorded default and the ladder counts in these docs in the same commit.
327
448
 
328
449
  ## Maintenance
329
450
 
330
- - Refresh the snapshot from the AA release page of each family
331
- (`/models/releases/<slug>`), the per-level model pages, and the harness
332
- comparison pages under `/agents/coding-agents/comparisons/`.
451
+ - Refresh each per-level row from the metric table of
452
+ `/models/comparisons/<level-slug>-vs-gpt-5-6-sol-high`. Check that the page
453
+ title names the requested model and level. Take `max` from the bare model
454
+ slug. Do not read values from the chart payloads of the model pages: each
455
+ chart holds only about 20 models, so a missing value there does not mean
456
+ that AA has not measured it.
457
+ - Refresh the Coding Agent Index from `/agents/coding-agents`.
333
458
  - Record the index version with the numbers. AA changes index composition
334
459
  between versions, so a score from another version is not a comparison.
335
460
  - Update the capture date in the same commit as any number.
@@ -11,6 +11,7 @@ import syncLayers from "../../docs/sync-layers.md" with { type: "text" };
11
11
  import install from "../../docs/install.md" with { type: "text" };
12
12
  import platforms from "../../docs/platforms.md" with { type: "text" };
13
13
  import ompModels from "../../docs/omp-models.md" with { type: "text" };
14
+ import ompContext from "../../docs/omp-context.md" with { type: "text" };
14
15
 
15
16
  const TOPICS: Record<string, { summary: string; body: string }> = {
16
17
  overview: { summary: "What docks-kit is and how the pieces fit", body: overview },
@@ -38,6 +39,10 @@ const TOPICS: Record<string, { summary: string; body: string }> = {
38
39
  summary: "omp role map and the Artificial Analysis snapshot behind it",
39
40
  body: ompModels,
40
41
  },
42
+ "omp-context": {
43
+ summary: "omp compaction trigger, the reserve-based default, and context-window lanes",
44
+ body: ompContext,
45
+ },
41
46
  };
42
47
 
43
48
  const topic = Argument.String("topic").pipe(
@@ -15,6 +15,7 @@ export const CODEX_REASONING_EFFORTS = [
15
15
  export const CLAUDE_ADVISOR_STATES = ["on", "off", "default"] as const;
16
16
 
17
17
  const VERIFIED = "2026-07-10";
18
+ const ADVISOR_VERIFIED = "2026-09-22";
18
19
  const DEFAULT = "default";
19
20
 
20
21
  const upstreamEfforts = (tool: Tool): ReadonlyArray<string> =>
@@ -80,8 +81,8 @@ export function effortCatalog(tool: Tool): string {
80
81
 
81
82
  export function advisorCatalog(): string {
82
83
  return [
83
- `Available claude advisor states (advisorModel; verified ${VERIFIED}):`,
84
- " on — set advisorModel: fable",
84
+ `Available claude advisor states (advisorModel; verified ${ADVISOR_VERIFIED}):`,
85
+ " on — set advisorModel: opus",
85
86
  " off — unset advisorModel",
86
87
  " default — SoT: off (unset)",
87
88
  ].join("\n");
@@ -101,15 +101,15 @@ export function syncClaudeAdvisor(ctx: Ctx, state: string): void {
101
101
  syncClaudeSetting(ctx, {
102
102
  tag: "--claude-advisor",
103
103
  key: "advisorModel",
104
- value: enabled ? "fable" : undefined,
105
- dryRun: enabled ? "set .advisorModel=fable" : "delete .advisorModel (advisor disabled)",
104
+ value: enabled ? "opus" : undefined,
105
+ dryRun: enabled ? "set .advisorModel=opus" : "delete .advisorModel (advisor disabled)",
106
106
  changed: enabled
107
- ? "Advisor: deployed settings advisorModel set to fable (SoT unchanged; flag-less sync reverts)"
107
+ ? "Advisor: deployed settings advisorModel set to opus (SoT unchanged; flag-less sync reverts)"
108
108
  : useDefault
109
109
  ? "Advisor: deployed settings advisorModel unset (SoT default: off)"
110
110
  : "Advisor: deployed settings advisorModel unset (--claude-advisor=off; SoT unchanged)",
111
111
  unchanged: enabled
112
- ? "Advisor: deployed settings advisorModel already fable"
112
+ ? "Advisor: deployed settings advisorModel already opus"
113
113
  : `Advisor: deployed settings advisorModel already unset (${useDefault ? "SoT default: off" : "advisor off"})`,
114
114
  });
115
115
  }