docks-kit 0.17.2 → 0.18.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +11 -6
- package/cli/docs/models.md +14 -14
- package/cli/docs/modifiers.md +4 -4
- package/cli/docs/omp-context.md +104 -0
- package/cli/docs/omp-models.md +269 -144
- package/cli/src/commands/docs.ts +5 -0
- package/cli/src/efforts.ts +3 -2
- package/cli/src/engine-native/claudeSettingsModifiers.ts +4 -4
- package/cli/src/engine-native/ompRemovals.ts +159 -0
- package/cli/src/engine-native/ompSync.ts +6 -2
- package/cli/src/generated/sotPayload.ts +8 -8
- package/docks-kit +13 -2
- package/docks-kit.ps1 +21 -6
- package/package.json +4 -3
package/cli/docs/omp-models.md
CHANGED
|
@@ -6,26 +6,56 @@ against.
|
|
|
6
6
|
|
|
7
7
|
## Role map
|
|
8
8
|
|
|
9
|
-
| Role | Model | Level | Index | Cost/task | TTFT |
|
|
10
|
-
|
|
11
|
-
| `default` | `anthropic/claude-opus-5` | high |
|
|
12
|
-
| `slow` | `anthropic/claude-opus-5` | xhigh |
|
|
13
|
-
| `plan` | `anthropic/claude-opus-5` | xhigh |
|
|
14
|
-
| `task` | `openai-codex/gpt-
|
|
15
|
-
| `advisor` | `openai-codex/gpt-
|
|
16
|
-
| `designer` | `anthropic/claude-opus-5` | high |
|
|
17
|
-
| `vision` | `anthropic/claude-opus-5` | medium |
|
|
18
|
-
| `smol` / `commit` | `openai-codex/gpt-
|
|
19
|
-
| `tiny` | `openai-codex/gpt-
|
|
20
|
-
| `fable` | `anthropic/claude-fable-5-1` | medium | 49 | $2.98 |
|
|
21
|
-
| `switch_fable` | `anthropic/claude-fable-5-1` | medium | 49 | $2.98 |
|
|
22
|
-
| `astra` | `openai-codex/gpt-6-astra` | xhigh |
|
|
9
|
+
| Role | Model | Level | Index | Cost/task | TTFT |
|
|
10
|
+
|---|---|---|---:|---:|---:|
|
|
11
|
+
| `default` | `anthropic/claude-opus-5-5` | high | 54 | $1.82 | 12.49 s |
|
|
12
|
+
| `slow` | `anthropic/claude-opus-5-5` | xhigh | 56 | $3.46 | 165.20 s |
|
|
13
|
+
| `plan` | `anthropic/claude-opus-5-5` | xhigh | 56 | $3.46 | 165.20 s |
|
|
14
|
+
| `task` | `openai-codex/gpt-6-sol` | high | 43 | $0.37 | n/a |
|
|
15
|
+
| `advisor` | `openai-codex/gpt-6-sol` | medium | 40 | $0.25 | n/a |
|
|
16
|
+
| `designer` | `anthropic/claude-opus-5-5` | high | 54 | $1.82 | 12.49 s |
|
|
17
|
+
| `vision` | `anthropic/claude-opus-5-5` | medium | 51 | $1.34 | 22.17 s |
|
|
18
|
+
| `smol` / `commit` | `openai-codex/gpt-6-luna` | medium | 29 | $0.02 | n/a |
|
|
19
|
+
| `tiny` | `openai-codex/gpt-6-luna` | low | 21 | $0.0045 | n/a |
|
|
20
|
+
| `fable` | `anthropic/claude-fable-5-1` | medium | 49 | $2.98 | 8.61 s |
|
|
21
|
+
| `switch_fable` | `anthropic/claude-fable-5-1` | medium | 49 | $2.98 | 8.61 s |
|
|
22
|
+
| `astra` | `openai-codex/gpt-6-astra` | xhigh | 52 | $2.31 | 188.20 s |
|
|
23
|
+
| `web` | `web/firecrawl` | n/a | n/a | n/a | n/a |
|
|
23
24
|
|
|
24
25
|
The table reports the measured Artificial Analysis figures for each assigned
|
|
25
26
|
model and level. It states no motive that the config or omp's own
|
|
26
|
-
documentation does not establish. AA
|
|
27
|
-
|
|
28
|
-
|
|
27
|
+
documentation does not establish. AA has not measured output speed or latency
|
|
28
|
+
for GPT-6 Sol or GPT-6 Luna at any level, so those rows carry `n/a` for TTFT.
|
|
29
|
+
AA measures no web search provider, so the `web` row carries no figures.
|
|
30
|
+
|
|
31
|
+
The role map carries no Coding Agent Index column. That index publishes one
|
|
32
|
+
entry per harness and model, and the Codex entries for GPT-6 Sol and GPT-6
|
|
33
|
+
Luna run at `max`, a level no role here uses. The snapshot section below
|
|
34
|
+
lists the entries.
|
|
35
|
+
|
|
36
|
+
### GPT-6 Sol and GPT-6 Luna availability
|
|
37
|
+
|
|
38
|
+
The `openai-codex` selectors for GPT-6 Sol and GPT-6 Luna were deployed on
|
|
39
|
+
2026-09-22, while the rollout of both models was still in progress. The owner
|
|
40
|
+
chose to deploy them before the rollout completed. Checks on that date, with
|
|
41
|
+
omp 18.2.9 and codex-cli 0.153.3:
|
|
42
|
+
|
|
43
|
+
- At 19:02 UTC, `codex exec -m gpt-6-sol` returned HTTP 400, `The 'gpt-6-sol'
|
|
44
|
+
model is not supported when using Codex with a ChatGPT account.`
|
|
45
|
+
`gpt-6-luna` returned the same error. The Codex model cache fetched at that
|
|
46
|
+
time listed neither model.
|
|
47
|
+
- At 19:14 UTC, the Codex model cache for the same account listed
|
|
48
|
+
`gpt-6-sol` and `gpt-6-luna`. A request on the new default could not be
|
|
49
|
+
tested, because the account had reached its Codex usage limit.
|
|
50
|
+
- The omp `openai-codex` catalog listed `gpt-6-astra` and the three `gpt-5.6`
|
|
51
|
+
models, but not `gpt-6-sol` or `gpt-6-luna`, both before and after
|
|
52
|
+
`omp models refresh` at 19:14 UTC. omp resolves a selector that is not in
|
|
53
|
+
the catalog by provider-scoped fuzzy match. An `omp -p --mode json` run on
|
|
54
|
+
`openai-codex/gpt-6-sol:low` recorded `gpt-5.6-sol` as the serving model,
|
|
55
|
+
and the Luna selector recorded `gpt-5.6-luna`. omp printed no warning.
|
|
56
|
+
|
|
57
|
+
Until the omp catalog lists both ids, the omp roles run GPT-5.6. To check, run
|
|
58
|
+
`omp models openai-codex`: the `gpt-6-sol` and `gpt-6-luna` rows must appear.
|
|
29
59
|
|
|
30
60
|
What omp's settings catalog establishes about these roles:
|
|
31
61
|
|
|
@@ -42,145 +72,235 @@ and `security-reviewer` exist as OMP agents (omp's task tool lists `scout`,
|
|
|
42
72
|
`reviewer`, `security-reviewer`, `task`, and `sonic`), so those two inherit
|
|
43
73
|
whatever `task` resolves to. The `code-reviewer` and `plan-reviewer` entries
|
|
44
74
|
are dormant until an OMP agent with that name exists.
|
|
45
|
-
`retry.fallbackChains.task` keeps `anthropic/claude-opus-5:high` as a
|
|
75
|
+
`retry.fallbackChains.task` keeps `anthropic/claude-opus-5-5:high` as a
|
|
46
76
|
cross-vendor fallback.
|
|
47
77
|
|
|
48
78
|
`retry.fallbackChains.astra` holds `anthropic/claude-fable-5-1:medium`.
|
|
49
79
|
`retry.fallbackChains.fable` holds `openai-codex/gpt-6-astra:xhigh`.
|
|
50
80
|
Each deliberate cycle stop falls to the other vendor. Without these explicit
|
|
51
81
|
chains, `retry.fallbackChains.default` would send either stop to
|
|
52
|
-
`openai-codex/gpt-
|
|
82
|
+
`openai-codex/gpt-6-sol:high`.
|
|
53
83
|
Chain entries are concrete selectors, not role aliases, so this pair cannot
|
|
54
84
|
recurse. The hidden `switch_fable` chain stays empty.
|
|
55
85
|
|
|
86
|
+
`modelRoles.web` is `web/firecrawl`, and `retry.fallbackChains.web` lists the
|
|
87
|
+
explicit 20-entry provider order that follows it. The two keys replace the
|
|
88
|
+
retired `providers.webSearchOrder` key, which omp no longer carries in its
|
|
89
|
+
settings schema. omp still accepts that key in a deployed file, expands it in
|
|
90
|
+
memory into the same two keys, and then drops it, but it never writes the
|
|
91
|
+
expansion back to disk. The kit therefore declares both keys itself.
|
|
92
|
+
|
|
93
|
+
The chain starts from the expansion omp produces, read back with
|
|
94
|
+
`omp config get retry.fallbackChains`. An explicit chain replaces omp's
|
|
95
|
+
built-in web order wholesale, so a shortened list drops providers instead of
|
|
96
|
+
reordering them. The kit makes two deliberate edits to omp's order:
|
|
97
|
+
|
|
98
|
+
- The Codex Luna entry follows the kit's Luna generation and names
|
|
99
|
+
`openai-codex/gpt-6-luna`.
|
|
100
|
+
- The owner removed every entry that named an older model:
|
|
101
|
+
`google/gemini-2.5-flash`, `google-antigravity/gemini-2.5-flash`,
|
|
102
|
+
`anthropic/claude-haiku-4-5`, `openai-codex/gpt-5.6`,
|
|
103
|
+
`openai-codex/gpt-5.5`, `xai/grok-4.5`, and `xai-oauth/grok-4.5`. omp never
|
|
104
|
+
tries those search backends now.
|
|
105
|
+
|
|
106
|
+
The order keeps Firecrawl, Exa, Perplexity, and Codex first. The remaining
|
|
107
|
+
`web/*` entries are omp's own ordering of the providers behind them.
|
|
108
|
+
|
|
56
109
|
## Artificial Analysis snapshot
|
|
57
110
|
|
|
58
|
-
Source: `https://artificialanalysis.ai`, read on 2026-09-
|
|
59
|
-
comes from one
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
111
|
+
Source: `https://artificialanalysis.ai`, read on 2026-09-22. Every figure below
|
|
112
|
+
comes from that one capture at Intelligence Index v4.3.2 and Coding Agent Index
|
|
113
|
+
v1.5. Each per-level row was read from the metric table of the comparison page
|
|
114
|
+
`/models/comparisons/<level-slug>-vs-gpt-5-6-sol-high`. The page title named
|
|
115
|
+
the requested model and level, and the page printed index v4.3.2. AA serves
|
|
116
|
+
the `max` level under the bare model slug. Index composition changed in v4.2,
|
|
117
|
+
again in v4.3, and again in v4.3.2, so figures from an earlier capture cannot
|
|
118
|
+
be mixed with these.
|
|
119
|
+
|
|
120
|
+
Coding Agent Index v1.5 (`/agents/coding-agents`) lists 12 harness and model
|
|
121
|
+
entries. Most carry `max`. Grok Build with Grok 4.7 runs at `xhigh`, and
|
|
122
|
+
Antigravity SDK with Gemini 3.8 Flash runs at `high`. Opencode with GLM-5.3
|
|
123
|
+
and Kimi Code CLI with Kimi K3 name no level. The entries that involve a model
|
|
124
|
+
family in this topic:
|
|
125
|
+
|
|
126
|
+
| Harness and model | Coding Agent Index |
|
|
127
|
+
|---|---:|
|
|
128
|
+
| Claude Code - Fable 5.1 (max, with fallback) | 62.2 |
|
|
129
|
+
| Codex - GPT-6 Astra (max) | 61.6 |
|
|
130
|
+
| Claude Code - Opus 5 (max) | 59.7 |
|
|
131
|
+
| Codex - GPT-6 Sol (max) | 56.7 |
|
|
132
|
+
| Codex - GPT-6 Luna (max) | 41.1 |
|
|
133
|
+
|
|
134
|
+
AA lists no Opus 5.5 entry. The page stores each score as a fraction, such as
|
|
135
|
+
`0.6222`, and this table shows it multiplied by 100.
|
|
66
136
|
|
|
67
137
|
Column meanings:
|
|
68
138
|
|
|
69
|
-
- **
|
|
70
|
-
Comparable only inside one index version.
|
|
71
|
-
- **
|
|
72
|
-
|
|
73
|
-
- **
|
|
74
|
-
|
|
75
|
-
- **Index output tokens** - total output tokens the model spends to complete the
|
|
139
|
+
- **Index** - AA Intelligence Index, a weighted aggregate across its
|
|
140
|
+
evaluation set. Comparable only inside one index version.
|
|
141
|
+
- **Cost/task** - weighted average USD to run one index task, including
|
|
142
|
+
input, cache, reasoning, and answer tokens.
|
|
143
|
+
- **Tokens/task** - answer plus reasoning tokens for one index task.
|
|
144
|
+
- **Index tokens** - total output tokens the model spends to complete the
|
|
76
145
|
whole index run. This is the token-efficiency signal.
|
|
146
|
+
- **Speed** - output tokens per second.
|
|
77
147
|
- **TTFT** - seconds to the first answer token, so reasoning time counts.
|
|
148
|
+
- **TB 4.0** - Terminal-Bench 4.0, one of the ten index evaluations.
|
|
149
|
+
|
|
150
|
+
### Claude Opus 5.5 (Anthropic) - `anthropic/claude-opus-5-5`
|
|
151
|
+
|
|
152
|
+
| Level | Index | Cost/task | Tokens/task | Index tokens | Speed t/s | TTFT s | TB 4.0 |
|
|
153
|
+
|---|---:|---:|---:|---:|---:|---:|---:|
|
|
154
|
+
| max | 58 | $5.98 | 119k | 260M | n/a | n/a | 60% |
|
|
155
|
+
| xhigh | 56 | $3.46 | 66k | 100M | 72 | 165.20 | 60% |
|
|
156
|
+
| high | 54 | $1.82 | 36k | 53M | 91 | 12.49 | 57% |
|
|
157
|
+
| medium | 51 | $1.34 | 26k | 38M | 76 | 22.17 | 53% |
|
|
158
|
+
| low | 42 | $0.55 | 10k | 20M | 94 | 4.79 | 31% |
|
|
159
|
+
|
|
160
|
+
Price: $4.00 in, $20.00 out, $0.20 cache hit per 1M. The Anthropic platform
|
|
161
|
+
documentation read the same day lists the same input, output, and cache-read
|
|
162
|
+
prices. It adds a $5.00 five-minute cache write and an $8.00 one-hour cache
|
|
163
|
+
write, which the AA comparison table does not show. Context 1M, maximum output
|
|
164
|
+
128K. Adaptive thinking is always on, and the Claude API default effort is
|
|
165
|
+
`medium`. All AA levels run with fallback. Max is the highest Intelligence
|
|
166
|
+
Index in this topic at this capture. AA has not measured speed or latency for
|
|
167
|
+
max. It measures a longer TTFT for medium than for high.
|
|
168
|
+
|
|
169
|
+
`SoT/.omp/models.yml` carries the Anthropic limits and prices as an `anthropic`
|
|
170
|
+
`modelOverrides` block, because the shared catalog still serves this id as a
|
|
171
|
+
stub with null limits and zero cost. Remove that block once the catalog
|
|
172
|
+
publishes the row.
|
|
78
173
|
|
|
79
|
-
###
|
|
174
|
+
### Claude Fable 5.1 (Anthropic) - `anthropic/claude-fable-5-1`
|
|
80
175
|
|
|
81
|
-
| Level |
|
|
82
|
-
|
|
83
|
-
| max | 53 |
|
|
84
|
-
| xhigh | 53 |
|
|
85
|
-
| high | 51 |
|
|
86
|
-
| medium |
|
|
87
|
-
| low |
|
|
88
|
-
| non-reasoning | 45 | n/a | $1.71 | 12M | n/a | n/a |
|
|
176
|
+
| Level | Index | Cost/task | Tokens/task | Index tokens | Speed t/s | TTFT s | TB 4.0 |
|
|
177
|
+
|---|---:|---:|---:|---:|---:|---:|---:|
|
|
178
|
+
| max | 53 | $7.63 | 78k | 188M | 66 | 311.51 | 52% |
|
|
179
|
+
| xhigh | 53 | $5.98 | 61k | 121M | 61 | 164.82 | 55% |
|
|
180
|
+
| high | 51 | $3.91 | 38k | 62M | 55 | 26.22 | 52% |
|
|
181
|
+
| medium | 49 | $2.98 | 28k | 44M | 55 | 8.61 | 45% |
|
|
182
|
+
| low | 47 | $2.37 | 22k | 33M | 54 | 7.28 | 40% |
|
|
89
183
|
|
|
90
|
-
Price: $10.00 in, $50.00 out, $
|
|
91
|
-
|
|
92
|
-
AA lists non-reasoning above low on cost per task.
|
|
184
|
+
Price: $10.00 in, $50.00 out, $0.25 cache hit per 1M. Context 1M. All levels
|
|
185
|
+
run with fallback.
|
|
93
186
|
|
|
94
|
-
###
|
|
187
|
+
### GPT-6 Astra (OpenAI) - `openai-codex/gpt-6-astra`
|
|
95
188
|
|
|
96
|
-
| Level |
|
|
97
|
-
|
|
98
|
-
| max |
|
|
99
|
-
| xhigh |
|
|
100
|
-
| high | 51 |
|
|
101
|
-
| medium |
|
|
102
|
-
| low |
|
|
103
|
-
|
|
104
|
-
Price: $10.00 in, $50.00 out, $
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
|
112
|
-
|
|
113
|
-
|
|
|
114
|
-
|
|
|
115
|
-
|
|
|
116
|
-
|
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
|
128
|
-
|
|
129
|
-
|
|
|
130
|
-
|
|
|
131
|
-
|
|
|
132
|
-
|
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
| Index
|
|
164
|
-
|
|
165
|
-
|
|
|
166
|
-
|
|
|
167
|
-
|
|
|
168
|
-
|
|
|
169
|
-
|
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
Astra
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
189
|
+
| Level | Index | Cost/task | Tokens/task | Index tokens | Speed t/s | TTFT s | TB 4.0 |
|
|
190
|
+
|---|---:|---:|---:|---:|---:|---:|---:|
|
|
191
|
+
| max | 53 | $3.26 | 27k | 60M | 61 | 322.65 | 59% |
|
|
192
|
+
| xhigh | 52 | $2.31 | 17k | 38M | 55 | 188.20 | 60% |
|
|
193
|
+
| high | 51 | $1.73 | 12k | 26M | 50 | 79.00 | 54% |
|
|
194
|
+
| medium | 50 | $1.54 | 10k | 19M | 48 | 6.19 | 49% |
|
|
195
|
+
| low | 46 | $0.82 | 4k | 10M | 51 | 2.76 | 42% |
|
|
196
|
+
|
|
197
|
+
Price: $10.00 in, $50.00 out, $1.00 cache hit per 1M. Context 1M. Knowledge
|
|
198
|
+
cutoff 2026-04-30. AA publishes no non-reasoning Astra row.
|
|
199
|
+
|
|
200
|
+
### GPT-6 Sol (OpenAI) - `openai-codex/gpt-6-sol`
|
|
201
|
+
|
|
202
|
+
| Level | Index | Cost/task | Tokens/task | Index tokens | Speed t/s | TTFT s | TB 4.0 |
|
|
203
|
+
|---|---:|---:|---:|---:|---:|---:|---:|
|
|
204
|
+
| max | 48 | $1.06 | 31k | 77M | n/a | n/a | 44% |
|
|
205
|
+
| xhigh | 44 | $0.53 | 16k | 40M | n/a | n/a | 30% |
|
|
206
|
+
| high | 43 | $0.37 | 10k | 25M | n/a | n/a | 26% |
|
|
207
|
+
| medium | 40 | $0.25 | 6k | 16M | n/a | n/a | 19% |
|
|
208
|
+
| low | 34 | $0.13 | 3k | 9M | n/a | n/a | 9% |
|
|
209
|
+
| non-reasoning | 28 | $0.33 | 5k | 8M | n/a | n/a | 13% |
|
|
210
|
+
|
|
211
|
+
Price: $2.00 in, $10.00 out, $0.20 cache hit per 1M. OpenAI's model page lists
|
|
212
|
+
a $2.50 cache write and a 1,050,000-token context with 128,000 maximum output
|
|
213
|
+
tokens. It bills a prompt above 272K input tokens at 2x input and cache rates
|
|
214
|
+
and 1.5x output for the full request. Knowledge cutoff 2026-04-20. The API
|
|
215
|
+
effort ladder is `none, low, medium, high, xhigh, max`, with `medium` as the
|
|
216
|
+
default. AA has not measured speed or latency for any level.
|
|
217
|
+
|
|
218
|
+
### GPT-6 Luna (OpenAI) - `openai-codex/gpt-6-luna`
|
|
219
|
+
|
|
220
|
+
| Level | Index | Cost/task | Tokens/task | Index tokens | Speed t/s | TTFT s | TB 4.0 |
|
|
221
|
+
|---|---:|---:|---:|---:|---:|---:|---:|
|
|
222
|
+
| max | 37 | $0.07 | 51k | 145M | n/a | n/a | 13% |
|
|
223
|
+
| xhigh | 34 | $0.04 | 27k | 68M | n/a | n/a | 8% |
|
|
224
|
+
| high | 32 | $0.03 | 20k | 47M | n/a | n/a | 5% |
|
|
225
|
+
| medium | 29 | $0.02 | 11k | 28M | n/a | n/a | 3% |
|
|
226
|
+
| low | 21 | $0.0045 | 2k | 8M | n/a | n/a | 0% |
|
|
227
|
+
| non-reasoning | 18 | $0.01 | 4k | 7M | n/a | n/a | 2% |
|
|
228
|
+
|
|
229
|
+
Price: $0.10 in, $0.50 out, $0.01 cache hit per 1M. OpenAI's model page lists
|
|
230
|
+
a $0.125 cache write, the same context, output, long-prompt billing, and
|
|
231
|
+
effort ladder as Sol, and a 2026-05-18 knowledge cutoff. AA has not measured
|
|
232
|
+
speed or latency for any level.
|
|
233
|
+
|
|
234
|
+
### GPT-5.6 Sol (OpenAI) - previous generation
|
|
235
|
+
|
|
236
|
+
The role map no longer uses GPT-5.6 Sol. This table stays as the measured
|
|
237
|
+
baseline for the GPT-6 Sol switch.
|
|
238
|
+
|
|
239
|
+
| Level | Index | Cost/task | Tokens/task | Index tokens | Speed t/s | TTFT s | TB 4.0 |
|
|
240
|
+
|---|---:|---:|---:|---:|---:|---:|---:|
|
|
241
|
+
| max | 47 | $1.99 | 29k | 90M | 82 | 130.17 | 40% |
|
|
242
|
+
| xhigh | 44 | $1.18 | 20k | 51M | 72 | 35.55 | 25% |
|
|
243
|
+
| high | 42 | $0.81 | 13k | 34M | 68 | 17.49 | 21% |
|
|
244
|
+
| medium | 39 | $0.50 | 8k | 21M | 58 | 5.07 | 15% |
|
|
245
|
+
| low | 33 | $0.26 | 4k | 13M | 58 | 3.85 | 1% |
|
|
246
|
+
| non-reasoning | 28 (estimated) | n/a | n/a | n/a | 64 | 1.22 | n/a |
|
|
247
|
+
|
|
248
|
+
Price: $4.00 in, $20.00 out, $0.40 cache hit per 1M. Context 1M. AA marks the
|
|
249
|
+
non-reasoning index score as estimated.
|
|
250
|
+
|
|
251
|
+
### GPT-5.6 Luna (OpenAI) - previous generation
|
|
252
|
+
|
|
253
|
+
The role map no longer uses GPT-5.6 Luna. This table stays as the measured
|
|
254
|
+
baseline for the GPT-6 Luna switch.
|
|
255
|
+
|
|
256
|
+
| Level | Index | Cost/task | Tokens/task | Index tokens | Speed t/s | TTFT s | TB 4.0 |
|
|
257
|
+
|---|---:|---:|---:|---:|---:|---:|---:|
|
|
258
|
+
| max | 37 | $0.18 | 41k | 154M | 145 | 122.15 | 12% |
|
|
259
|
+
| xhigh | 35 | $0.09 | 24k | 85M | 143 | 40.24 | 4% |
|
|
260
|
+
| high | 32 | $0.04 | 14k | 50M | 132 | 15.15 | 3% |
|
|
261
|
+
| medium | 25 | $0.02 | 4k | 18M | 133 | 2.57 | 1% |
|
|
262
|
+
| low | 21 | $0.01 | 3k | 10M | 131 | 1.69 | 0% |
|
|
263
|
+
| non-reasoning | 16 | $0.01 | 2k | 5M | 140 | 0.81 | 1% |
|
|
264
|
+
|
|
265
|
+
Price: $0.20 in, $1.20 out, $0.02 cache hit per 1M. Context 1M.
|
|
266
|
+
|
|
267
|
+
## Why `task` runs GPT-6 Sol high
|
|
268
|
+
|
|
269
|
+
`task` runs `openai-codex/gpt-6-sol:high`, the same level GPT-5.6 Sol ran
|
|
270
|
+
before it. The owner uses Astra only for main orchestration, so Astra has a
|
|
271
|
+
dedicated `astra` cycle stop at `xhigh`. The comparison below records the
|
|
272
|
+
retired Astra-low and GPT-5.6 Sol choices beside the current GPT-6 Sol choice.
|
|
273
|
+
|
|
274
|
+
| Metric | Astra low, retired | GPT-5.6 Sol high, previous | GPT-6 Sol high, current | GPT-6 Sol max | Opus 5.5 high, `default` |
|
|
275
|
+
|---|---:|---:|---:|---:|---:|
|
|
276
|
+
| Intelligence Index | 46 | 42 | 43 | 48 | 54 |
|
|
277
|
+
| Cost per Index task | $0.82 | $0.81 | $0.37 | $1.06 | $1.82 |
|
|
278
|
+
| Output tokens per task | 4k | 13k | 10k | 31k | 36k |
|
|
279
|
+
| Index output tokens | 10M | 34M | 25M | 77M | 53M |
|
|
280
|
+
| Answer TTFT | 2.76 s | 17.49 s | n/a | n/a | 12.49 s |
|
|
281
|
+
| End-to-end response time | 12.49 s | 24.87 s | n/a | n/a | 18.00 s |
|
|
282
|
+
| Time per index task | 87.88 s | 189.16 s | n/a | n/a | 244.35 s |
|
|
283
|
+
| Terminal-Bench 4.0 | 42% | 21% | 26% | 44% | 57% |
|
|
284
|
+
| AA-Briefcase v1.1 | 1261 | 1370 | 1289 | 1483 | 1705 |
|
|
285
|
+
| AA-Omniscience | 41 | 20 | 27 | 27 | 41 |
|
|
286
|
+
|
|
287
|
+
GPT-6 Sol high scores one index point above GPT-5.6 Sol high. It costs $0.37
|
|
288
|
+
against $0.81 per index task and uses 10k output tokens per task against 13k.
|
|
289
|
+
It also scores 26% against 21% on Terminal-Bench 4.0. AA-Briefcase is the one
|
|
290
|
+
metric where the older model leads, at 1370 against 1289. AA has not measured
|
|
291
|
+
GPT-6 Sol latency, so the latency comparison is open.
|
|
292
|
+
|
|
293
|
+
Astra low still beats GPT-6 Sol high on the index, at 46 against 43, and on
|
|
294
|
+
Terminal-Bench 4.0, at 42% against 26%. It costs $0.82 against $0.37 per index
|
|
295
|
+
task. The owner reserves Astra for interactive orchestration.
|
|
296
|
+
|
|
297
|
+
Opus 5.5 high scores 11 points above GPT-6 Sol high and costs $1.82 against
|
|
298
|
+
$0.37 per index task. It is the `default`, `designer`, and fallback model, not
|
|
299
|
+
the `task` model, so the subagent fan-out keeps the cheaper Sol.
|
|
300
|
+
|
|
301
|
+
The `astra` cycle stop runs xhigh: index 52, $2.31 per index task, and
|
|
302
|
+
188.20 s TTFT. AA publishes a Coding Agent Index entry for Astra only at max,
|
|
303
|
+
paired with Codex, where it scores 61.6.
|
|
184
304
|
|
|
185
305
|
## How Astra is selected in practice
|
|
186
306
|
|
|
@@ -201,16 +321,17 @@ quick answer.
|
|
|
201
321
|
not Sol or Astra. To move them, change `modelRoles.smol` or add a
|
|
202
322
|
`task.agentModelOverrides` entry for the agent name.
|
|
203
323
|
- The bundled `task` agent carries `model: "@task"` and
|
|
204
|
-
`thinking-level: auto`. It resolves Sol, and `auto` classifies each prompt
|
|
324
|
+
`thinking-level: auto`. It resolves GPT-6 Sol, and `auto` classifies each prompt
|
|
205
325
|
to choose a thinking level.
|
|
206
326
|
- `task.enableEffort` is `true`, so a caller can pass `effort: lo`, `med`, or
|
|
207
327
|
`hi`, which overrides `auto`.
|
|
208
|
-
- `task.maxEffort` is `
|
|
209
|
-
default and Luna `
|
|
210
|
-
- The bundled `reviewer` and `security-reviewer` inherit `@task`, now
|
|
328
|
+
- `task.maxEffort` is `max`, so `scout` and `sonic` run GPT-6 Luna `medium`
|
|
329
|
+
by default and GPT-6 Luna `max` with `effort: hi`.
|
|
330
|
+
- The bundled `reviewer` and `security-reviewer` inherit `@task`, now GPT-6
|
|
331
|
+
Sol high.
|
|
211
332
|
- The `code-reviewer` and `plan-reviewer` override entries remain dormant.
|
|
212
|
-
Both point to `@task`, now Sol high. omp's task tool rejects both
|
|
213
|
-
unknown agents, so neither can spawn.
|
|
333
|
+
Both point to `@task`, now GPT-6 Sol high. omp's task tool rejects both
|
|
334
|
+
names as unknown agents, so neither can spawn.
|
|
214
335
|
|
|
215
336
|
The runtime per-agent measurement from fresh `omp -p` runs on 2026-09-09
|
|
216
337
|
predates this change. It applies to the retired Astra-low `task` configuration,
|
|
@@ -327,9 +448,13 @@ recorded default and the ladder counts in these docs in the same commit.
|
|
|
327
448
|
|
|
328
449
|
## Maintenance
|
|
329
450
|
|
|
330
|
-
- Refresh
|
|
331
|
-
|
|
332
|
-
|
|
451
|
+
- Refresh each per-level row from the metric table of
|
|
452
|
+
`/models/comparisons/<level-slug>-vs-gpt-5-6-sol-high`. Check that the page
|
|
453
|
+
title names the requested model and level. Take `max` from the bare model
|
|
454
|
+
slug. Do not read values from the chart payloads of the model pages: each
|
|
455
|
+
chart holds only about 20 models, so a missing value there does not mean
|
|
456
|
+
that AA has not measured it.
|
|
457
|
+
- Refresh the Coding Agent Index from `/agents/coding-agents`.
|
|
333
458
|
- Record the index version with the numbers. AA changes index composition
|
|
334
459
|
between versions, so a score from another version is not a comparison.
|
|
335
460
|
- Update the capture date in the same commit as any number.
|
package/cli/src/commands/docs.ts
CHANGED
|
@@ -11,6 +11,7 @@ import syncLayers from "../../docs/sync-layers.md" with { type: "text" };
|
|
|
11
11
|
import install from "../../docs/install.md" with { type: "text" };
|
|
12
12
|
import platforms from "../../docs/platforms.md" with { type: "text" };
|
|
13
13
|
import ompModels from "../../docs/omp-models.md" with { type: "text" };
|
|
14
|
+
import ompContext from "../../docs/omp-context.md" with { type: "text" };
|
|
14
15
|
|
|
15
16
|
const TOPICS: Record<string, { summary: string; body: string }> = {
|
|
16
17
|
overview: { summary: "What docks-kit is and how the pieces fit", body: overview },
|
|
@@ -38,6 +39,10 @@ const TOPICS: Record<string, { summary: string; body: string }> = {
|
|
|
38
39
|
summary: "omp role map and the Artificial Analysis snapshot behind it",
|
|
39
40
|
body: ompModels,
|
|
40
41
|
},
|
|
42
|
+
"omp-context": {
|
|
43
|
+
summary: "omp compaction trigger, the reserve-based default, and context-window lanes",
|
|
44
|
+
body: ompContext,
|
|
45
|
+
},
|
|
41
46
|
};
|
|
42
47
|
|
|
43
48
|
const topic = Argument.String("topic").pipe(
|
package/cli/src/efforts.ts
CHANGED
|
@@ -15,6 +15,7 @@ export const CODEX_REASONING_EFFORTS = [
|
|
|
15
15
|
export const CLAUDE_ADVISOR_STATES = ["on", "off", "default"] as const;
|
|
16
16
|
|
|
17
17
|
const VERIFIED = "2026-07-10";
|
|
18
|
+
const ADVISOR_VERIFIED = "2026-09-22";
|
|
18
19
|
const DEFAULT = "default";
|
|
19
20
|
|
|
20
21
|
const upstreamEfforts = (tool: Tool): ReadonlyArray<string> =>
|
|
@@ -80,8 +81,8 @@ export function effortCatalog(tool: Tool): string {
|
|
|
80
81
|
|
|
81
82
|
export function advisorCatalog(): string {
|
|
82
83
|
return [
|
|
83
|
-
`Available claude advisor states (advisorModel; verified ${
|
|
84
|
-
" on — set advisorModel:
|
|
84
|
+
`Available claude advisor states (advisorModel; verified ${ADVISOR_VERIFIED}):`,
|
|
85
|
+
" on — set advisorModel: opus",
|
|
85
86
|
" off — unset advisorModel",
|
|
86
87
|
" default — SoT: off (unset)",
|
|
87
88
|
].join("\n");
|
|
@@ -101,15 +101,15 @@ export function syncClaudeAdvisor(ctx: Ctx, state: string): void {
|
|
|
101
101
|
syncClaudeSetting(ctx, {
|
|
102
102
|
tag: "--claude-advisor",
|
|
103
103
|
key: "advisorModel",
|
|
104
|
-
value: enabled ? "
|
|
105
|
-
dryRun: enabled ? "set .advisorModel=
|
|
104
|
+
value: enabled ? "opus" : undefined,
|
|
105
|
+
dryRun: enabled ? "set .advisorModel=opus" : "delete .advisorModel (advisor disabled)",
|
|
106
106
|
changed: enabled
|
|
107
|
-
? "Advisor: deployed settings advisorModel set to
|
|
107
|
+
? "Advisor: deployed settings advisorModel set to opus (SoT unchanged; flag-less sync reverts)"
|
|
108
108
|
: useDefault
|
|
109
109
|
? "Advisor: deployed settings advisorModel unset (SoT default: off)"
|
|
110
110
|
: "Advisor: deployed settings advisorModel unset (--claude-advisor=off; SoT unchanged)",
|
|
111
111
|
unchanged: enabled
|
|
112
|
-
? "Advisor: deployed settings advisorModel already
|
|
112
|
+
? "Advisor: deployed settings advisorModel already opus"
|
|
113
113
|
: `Advisor: deployed settings advisorModel already unset (${useDefault ? "SoT default: off" : "advisor off"})`,
|
|
114
114
|
});
|
|
115
115
|
}
|