@ai-outfitter/outfitter 1.8.1 → 1.10.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/dist/agents/ClaudeConfigStrategy.d.ts +15 -0
- package/dist/agents/ClaudeConfigStrategy.js +53 -0
- package/dist/agents/ClaudeConfigStrategy.js.map +1 -0
- package/dist/cli/commands/RunAgentCommand.d.ts +5 -0
- package/dist/cli/commands/RunAgentCommand.js +30 -8
- package/dist/cli/commands/RunAgentCommand.js.map +1 -1
- package/dist/projection/Materialize.d.ts +7 -0
- package/dist/projection/Materialize.js +24 -0
- package/dist/projection/Materialize.js.map +1 -1
- package/dist/projection/ProjectHarness.js +48 -10
- package/dist/projection/ProjectHarness.js.map +1 -1
- package/dist/projection/Projection.d.ts +9 -1
- package/dist/schemas/agent.schema.json +13 -4
- package/dist/schemas/settings.schema.json +4 -0
- package/dist/settings/Settings.d.ts +14 -0
- package/dist/settings/Settings.js +7 -0
- package/dist/settings/Settings.js.map +1 -1
- package/dist/settings/SettingsLoader.js +4 -2
- package/dist/settings/SettingsLoader.js.map +1 -1
- package/dist/settings/SettingsMerger.js +3 -0
- package/dist/settings/SettingsMerger.js.map +1 -1
- package/dist/setup/Setup.js +7 -4
- package/dist/setup/Setup.js.map +1 -1
- package/docs/documentation/README.md +2 -0
- package/docs/documentation/cli.md +8 -6
- package/docs/documentation/cost-estimation.md +466 -0
- package/docs/documentation/porting-claude.md +4 -0
- package/docs/documentation/settings.md +2 -0
- package/docs/documentation/state.md +17 -6
- package/docs/documentation/support-matrix.md +5 -4
- package/package.json +1 -1
- package/src/schemas/agent.schema.json +13 -4
- package/src/schemas/settings.schema.json +4 -0
|
@@ -0,0 +1,466 @@
|
|
|
1
|
+
# Estimate agent workload cost
|
|
2
|
+
|
|
3
|
+
This table estimates what one agent that never stops costs in model tokens.
|
|
4
|
+
Every figure comes from a dated price book and an estimated token rate, not
|
|
5
|
+
from an invoice. The day column is 24 hours at the stated rate. The month
|
|
6
|
+
column is 730 hours, the average month.
|
|
7
|
+
|
|
8
|
+
| Provider | Model | Steady: day / month | Saturated: day / month |
|
|
9
|
+
| --------- | ---------------- | ------------------: | ---------------------: |
|
|
10
|
+
| Anthropic | Claude Fable 5 | $182.40 / $5,548 | $408.00 / $12,410 |
|
|
11
|
+
| OpenAI | GPT-5.6 Sol | $98.40 / $2,993 | $216.00 / $6,570 |
|
|
12
|
+
| Anthropic | Claude Opus 5 | $91.20 / $2,774 | $204.00 / $6,205 |
|
|
13
|
+
| OpenAI | GPT-5.6 Terra | $39.36 / $1,197 | $86.40 / $2,628 |
|
|
14
|
+
| Anthropic | Claude Sonnet 5 | $36.48 / $1,110 | $81.60 / $2,482 |
|
|
15
|
+
| Anthropic | Claude Haiku 4.5 | $18.24 / $555 | $40.80 / $1,241 |
|
|
16
|
+
| OpenAI | GPT-5.4 mini | $14.76 / $449 | $32.40 / $986 |
|
|
17
|
+
| OpenAI | GPT-5.6 Luna | $3.94 / $120 | $8.64 / $263 |
|
|
18
|
+
|
|
19
|
+
Four notes apply to every row:
|
|
20
|
+
|
|
21
|
+
- **The two rates.** Steady is ordinary agent turns with repeated instructions:
|
|
22
|
+
400K input, 600K cache-read, and 60K output tokens for each active hour.
|
|
23
|
+
Saturated is continuous agentic coding with large tool results:
|
|
24
|
+
1M / 2M / 100K.
|
|
25
|
+
- **Scope.** This is the model line only. It excludes substrate, tools,
|
|
26
|
+
subagents, and cache writes. See
|
|
27
|
+
[What this figure excludes](#what-this-figure-excludes).
|
|
28
|
+
- **Luna reserve.** Apply a five-times reserve to GPT-5.6 Luna until an invoice
|
|
29
|
+
settles its disputed price. That gives $19.68 / $599 steady and
|
|
30
|
+
$43.20 / $1,314 saturated.
|
|
31
|
+
- **Part-time agents.** Cost scales with active hours. Eight hours on each
|
|
32
|
+
working day is about 176 hours, which is 24 percent of the month column.
|
|
33
|
+
|
|
34
|
+
At the same token rate, GPT-5.6 Sol costs about 25 times GPT-5.6 Luna and
|
|
35
|
+
Claude Fable 5 costs about 46 times. Apply the Luna reserve and those spreads
|
|
36
|
+
fall to 5 and 9 times. Against Claude Haiku 4.5, the cheapest model with an
|
|
37
|
+
undisputed price, they are 5.4 and 10 times. The two rates change the cost of
|
|
38
|
+
one model by about 2.2 times. Model choice and work intensity are separate
|
|
39
|
+
controls.
|
|
40
|
+
|
|
41
|
+
This guide calculates a planning estimate. It does not read invoices or collect
|
|
42
|
+
usage. See
|
|
43
|
+
[Replace estimates with measurements](#replace-estimates-with-measurements) for
|
|
44
|
+
the record that can replace each estimate later.
|
|
45
|
+
|
|
46
|
+
## Price book
|
|
47
|
+
|
|
48
|
+
Two providers, two capture dates. Prices are United States dollars for one
|
|
49
|
+
million text tokens.
|
|
50
|
+
|
|
51
|
+
OpenAI standard text-token prices, captured 2026-08-15 from the
|
|
52
|
+
[OpenAI model documentation](https://developers.openai.com/api/docs/models):
|
|
53
|
+
|
|
54
|
+
| Model | Documented role | Input | Cache read | Output |
|
|
55
|
+
| ------------- | ----------------------------------------------- | ----: | ---------: | -----: |
|
|
56
|
+
| GPT-5.6 Sol | Frontier tier | $5.00 | $0.50 | $30.00 |
|
|
57
|
+
| GPT-5.6 Terra | Balance of intelligence and cost | $2.00 | $0.20 | $12.00 |
|
|
58
|
+
| GPT-5.4 mini | High-volume coding, computer use, and subagents | $0.75 | $0.075 | $4.50 |
|
|
59
|
+
| GPT-5.6 Luna | Cost-sensitive, high-volume tier | $0.20 | $0.02 | $1.20 |
|
|
60
|
+
|
|
61
|
+
Anthropic standard text-token prices, captured 2026-08-17 from the
|
|
62
|
+
[Claude platform pricing page](https://platform.claude.com/docs/en/about-claude/pricing):
|
|
63
|
+
|
|
64
|
+
| Model | Documented role | Input | Cache read | Output |
|
|
65
|
+
| ---------------- | ------------------------------------------ | -----: | ---------: | -----: |
|
|
66
|
+
| Claude Fable 5 | Demanding reasoning and long-horizon work | $10.00 | $1.00 | $50.00 |
|
|
67
|
+
| Claude Opus 5 | Complex agentic coding and enterprise work | $5.00 | $0.50 | $25.00 |
|
|
68
|
+
| Claude Sonnet 5 | Balance of speed and intelligence | $2.00 | $0.20 | $10.00 |
|
|
69
|
+
| Claude Haiku 4.5 | Fastest and most cost-effective | $1.00 | $0.10 | $5.00 |
|
|
70
|
+
|
|
71
|
+
Three properties of these tables control every figure in this guide:
|
|
72
|
+
|
|
73
|
+
- The Anthropic price vectors are exactly proportional on every token class.
|
|
74
|
+
`Fable 5 : Opus 5 : Sonnet 5 : Haiku 4.5 = 10 : 5 : 2 : 1`. Therefore any
|
|
75
|
+
Anthropic figure in this guide scales by that ratio, whatever the token mix
|
|
76
|
+
is. Use this property to check the arithmetic.
|
|
77
|
+
- **Neither table contains a cache-write price, and both providers charge
|
|
78
|
+
one.** An OpenAI explicit cache write costs 1.25 times the uncached-input
|
|
79
|
+
price. An Anthropic cache write costs 1.25 times input for a 5-minute
|
|
80
|
+
lifetime and 2 times input for a 1-hour lifetime. Every figure in this guide
|
|
81
|
+
omits that line, so every figure understates the cost. See
|
|
82
|
+
[Correction from measured usage](#correction-from-measured-usage) for the
|
|
83
|
+
size of that error.
|
|
84
|
+
- **The GPT-5.6 Luna price is disputed.** The 2026-08-15 capture and a later
|
|
85
|
+
live re-check of the model catalog, the comparison page, and the Luna page
|
|
86
|
+
all gave $0.20 / $0.02 / $1.20. Official-domain search results crawled two
|
|
87
|
+
weeks earlier gave $1.00 / $0.10 / $6.00, which is five times more. No
|
|
88
|
+
invoice has settled it. That ratio is the five-times reserve this guide
|
|
89
|
+
applies to every Luna figure.
|
|
90
|
+
|
|
91
|
+
In the 2026-08-17 capture, the Claude Sonnet 5 introductory price of
|
|
92
|
+
$2.00 / $10.00 is the standard price, and a previously scheduled increase to
|
|
93
|
+
$3.00 / $15.00 is no longer listed. An estimate built on the increase is about
|
|
94
|
+
50 percent too high.
|
|
95
|
+
|
|
96
|
+
## How this figure is calculated
|
|
97
|
+
|
|
98
|
+
Three inputs produce every number in the table above: a dated price book, an
|
|
99
|
+
estimated token rate, and the active hours in the period.
|
|
100
|
+
|
|
101
|
+
### Step 1 — select a token rate
|
|
102
|
+
|
|
103
|
+
The rate is the only estimated input. Two rates bound realistic agent work.
|
|
104
|
+
Both come from the workload profile, not from a measured ledger:
|
|
105
|
+
|
|
106
|
+
| Rate | Per active hour: input / cache read / output | Derivation |
|
|
107
|
+
| --------- | -------------------------------------------- | ---------------------------------------------------------------- |
|
|
108
|
+
| Steady | 400K / 600K / 60K | 100K / 150K / 15K for each run, 15 minutes each run, 4 each hour |
|
|
109
|
+
| Saturated | 1M / 2M / 100K | 500K / 1M / 50K for each run, 30 minutes each run, 2 each hour |
|
|
110
|
+
|
|
111
|
+
No observed provider record supplies these two rates. Every dollar figure in
|
|
112
|
+
this guide inherits that uncertainty.
|
|
113
|
+
|
|
114
|
+
### Step 2 — calculate the cost of one active agent-hour
|
|
115
|
+
|
|
116
|
+
Apply the price book to the selected rate. For Claude Opus 5 at the steady
|
|
117
|
+
rate:
|
|
118
|
+
|
|
119
|
+
```text
|
|
120
|
+
400,000 input × $5.00 / 1,000,000 = $2.00
|
|
121
|
+
+ 600,000 cache read × $0.50 / 1,000,000 = $0.30
|
|
122
|
+
+ 60,000 output × $25.00 / 1,000,000 = $1.50
|
|
123
|
+
-----
|
|
124
|
+
$3.80 per active agent-hour
|
|
125
|
+
```
|
|
126
|
+
|
|
127
|
+
The general form, with cache writes and reasoning, is in
|
|
128
|
+
[Calculate model cost once](#calculate-model-cost-once). The same calculation
|
|
129
|
+
for every model:
|
|
130
|
+
|
|
131
|
+
| Model | Steady | Saturated |
|
|
132
|
+
| ---------------- | -----: | --------: |
|
|
133
|
+
| Claude Fable 5 | $7.60 | $17.00 |
|
|
134
|
+
| GPT-5.6 Sol | $4.10 | $9.00 |
|
|
135
|
+
| Claude Opus 5 | $3.80 | $8.50 |
|
|
136
|
+
| GPT-5.6 Terra | $1.64 | $3.60 |
|
|
137
|
+
| Claude Sonnet 5 | $1.52 | $3.40 |
|
|
138
|
+
| Claude Haiku 4.5 | $0.76 | $1.70 |
|
|
139
|
+
| GPT-5.4 mini | $0.615 | $1.35 |
|
|
140
|
+
| GPT-5.6 Luna | $0.164 | $0.36 |
|
|
141
|
+
|
|
142
|
+
### Step 3 — multiply by the active hours in the period
|
|
143
|
+
|
|
144
|
+
A non-stop agent is active for 24 hours on each day and 730 hours in each
|
|
145
|
+
month. For Claude Opus 5 at the steady rate:
|
|
146
|
+
|
|
147
|
+
```text
|
|
148
|
+
$3.80 × 24 = $91.20 for one day
|
|
149
|
+
$3.80 × 730 = $2,774 for one month
|
|
150
|
+
```
|
|
151
|
+
|
|
152
|
+
The table at the top of this guide applies these two multiplications to every
|
|
153
|
+
model.
|
|
154
|
+
|
|
155
|
+
## What this figure excludes
|
|
156
|
+
|
|
157
|
+
The table at the top of this guide is the model line only. The full equation
|
|
158
|
+
has five more terms. The prices quoted below come from the same 2026-08-17
|
|
159
|
+
Anthropic capture as the price book.
|
|
160
|
+
|
|
161
|
+
- **Cache writes.** See the price book above. This is the one excluded term
|
|
162
|
+
with a measured size. See
|
|
163
|
+
[Correction from measured usage](#correction-from-measured-usage).
|
|
164
|
+
- **Substrate.** A non-stop agent holds compute for all 730 hours. Add the
|
|
165
|
+
hourly price of the node or pod. Claude Managed Agents prices this line
|
|
166
|
+
directly at $0.08 for each session-hour, which is $58.40 for a non-stop
|
|
167
|
+
month.
|
|
168
|
+
- **Tool fees.** Anthropic web search costs $10 for 1,000 searches. An agent
|
|
169
|
+
that searches once each minute non-stop adds $438 each month.
|
|
170
|
+
- **Subagents.** A coordinator that runs subagents multiplies the token rate by
|
|
171
|
+
its concurrency. These rates describe one serial thread.
|
|
172
|
+
- **Long-context and residency uplifts.** A GPT-5.6 request above 272,000 input
|
|
173
|
+
tokens can cost more. Anthropic `inference_geo: "us"` adds 1.1 times.
|
|
174
|
+
|
|
175
|
+
## Correction from measured usage
|
|
176
|
+
|
|
177
|
+
One measured record exists, and it corrects the token mix that the estimate
|
|
178
|
+
assumes.
|
|
179
|
+
|
|
180
|
+
A single workstation retained 2,804 Claude Code session files covering 42
|
|
181
|
+
active days. Each assistant turn carries a `usage` object. Aggregating all
|
|
182
|
+
221,609 turns gives 45.84 billion tokens with this composition:
|
|
183
|
+
|
|
184
|
+
| Token class | Assumed by the rates | Measured |
|
|
185
|
+
| -------------- | -------------------: | -------: |
|
|
186
|
+
| Uncached input | ~38% | 0.04% |
|
|
187
|
+
| Cache read | ~57% | 96.76% |
|
|
188
|
+
| Cache write | 0%, omitted | 2.82% |
|
|
189
|
+
| Output | ~6% | 0.39% |
|
|
190
|
+
|
|
191
|
+
Two corrections follow:
|
|
192
|
+
|
|
193
|
+
1. **Uncached input is a rounding error.** Real agent work re-reads a cached
|
|
194
|
+
prefix on almost every turn. A cost model weighted toward uncached input
|
|
195
|
+
prices the wrong thing. Priced on Claude Opus 5, the measured mix costs
|
|
196
|
+
about 4.7 times less for each token than the steady rate does.
|
|
197
|
+
2. **Cache writes are the second-largest cost line.** The 125,394 Claude Opus 5
|
|
198
|
+
turns in this record used 24.51B cache-read, 703M cache-write, 86M output,
|
|
199
|
+
and 2.08M uncached-input tokens. Priced on Claude Opus 5 and treating every
|
|
200
|
+
cache write as a 5-minute write, that is $12,255 cache read, $4,397 cache
|
|
201
|
+
write, $2,151 output, and $10 uncached input: 65, 23, 11, and 0.05 percent.
|
|
202
|
+
A table that omits a line of that size reports about 77 percent of the model
|
|
203
|
+
cost.
|
|
204
|
+
|
|
205
|
+
Correction 2 is worth about 1.3 times, and correction 1 about 4.7 times. Both
|
|
206
|
+
are smaller than the 25-times spread between model tiers, so model choice
|
|
207
|
+
remains the larger control. Neither correction transfers directly to the tables
|
|
208
|
+
above: the two rates assume no cache writes at all, so their cache-write error
|
|
209
|
+
is unknown rather than 23 percent.
|
|
210
|
+
|
|
211
|
+
This record measures one human driving agents interactively between 2026-07-05
|
|
212
|
+
and 2026-08-17. It does not measure a resident Deployment, and it is not a
|
|
213
|
+
provider invoice. Treat it as evidence about token composition, not as a price
|
|
214
|
+
for the non-stop case.
|
|
215
|
+
|
|
216
|
+
## Evidence class for each number
|
|
217
|
+
|
|
218
|
+
| Value | Class | Basis |
|
|
219
|
+
| --------------------------- | ---------- | --------------------------------------------------------------- |
|
|
220
|
+
| Model prices | Price book | Dated public captures, 2026-08-15 and 2026-08-17 |
|
|
221
|
+
| 730 hours in a month | Convention | The [`730 / interval_hours` rule](#define-one-workload-profile) |
|
|
222
|
+
| The two token rates | Estimated | Workload profile assumptions, no ledger |
|
|
223
|
+
| Cost per hour and per month | Derived | Price book times the estimated rate |
|
|
224
|
+
| Token-class composition | Measured | 221,609 assistant turns over 42 days |
|
|
225
|
+
| GPT-5.6 Luna price | Disputed | $0.20 / $0.02 / $1.20 against $1.00 / $0.10 / $6.00, no invoice |
|
|
226
|
+
| Substrate and tool prices | Price book | The 2026-08-17 Anthropic capture |
|
|
227
|
+
| Every other excluded term | Unknown | Not zero |
|
|
228
|
+
|
|
229
|
+
A price-book calculation is not an observed provider charge. An unknown value
|
|
230
|
+
is not zero.
|
|
231
|
+
|
|
232
|
+
## Cost boundary
|
|
233
|
+
|
|
234
|
+
Use one workload profile for every execution surface. Change only the substrate
|
|
235
|
+
inputs when you compare a local run, GitHub Actions, a scheduled Kubernetes
|
|
236
|
+
Job, or a resident Deployment. Use this equation:
|
|
237
|
+
|
|
238
|
+
```text
|
|
239
|
+
monthly cost =
|
|
240
|
+
fixed substrate cost
|
|
241
|
+
+ runs × (model + tools + variable compute)
|
|
242
|
+
+ storage
|
|
243
|
+
+ network
|
|
244
|
+
```
|
|
245
|
+
|
|
246
|
+
The terms mean:
|
|
247
|
+
|
|
248
|
+
| Term | Include |
|
|
249
|
+
| ---------------- | ------------------------------------------------------------------------------------- |
|
|
250
|
+
| Fixed substrate | Compute that accrues while no run is active. A resident Deployment is the usual case. |
|
|
251
|
+
| Runs | Every initial attempt, retry, and delegated run in the month. |
|
|
252
|
+
| Model | Input, cache-read, cache-write, output, and reasoning charges for one run. |
|
|
253
|
+
| Tools | Search, browser, image, sandbox, or other billable tool use for one run. |
|
|
254
|
+
| Variable compute | Runner minutes or Job resources that accrue only while a run is active. |
|
|
255
|
+
| Storage | Session data, artifacts, logs, volumes, and backups. |
|
|
256
|
+
| Network | Egress, gateways, and other billed transfer. |
|
|
257
|
+
|
|
258
|
+
An included quota changes the billed amount. It does not change usage. Record
|
|
259
|
+
the usage before you apply the quota.
|
|
260
|
+
|
|
261
|
+
## Define one workload profile
|
|
262
|
+
|
|
263
|
+
Enter these values once:
|
|
264
|
+
|
|
265
|
+
| Input | Unit | Rule |
|
|
266
|
+
| ------------------------------- | ---------------- | ------------------------------------------------------------------ |
|
|
267
|
+
| Workload identity | stable text | Use the same identity on every surface. |
|
|
268
|
+
| Interval | hours | Use the time between scheduled starts. |
|
|
269
|
+
| Attempts per scheduled start | count | Include the first attempt and expected retries. |
|
|
270
|
+
| Delegated runs per attempt | count | Count each subagent or delegated Job separately. |
|
|
271
|
+
| Input tokens | tokens per run | Keep uncached input separate from cache reads. |
|
|
272
|
+
| Cache-read tokens | tokens per run | Use `null`, not zero, when the provider does not report them. |
|
|
273
|
+
| Cache-write tokens | tokens per run | Use `null`, not zero, when the provider does not report them. |
|
|
274
|
+
| Output tokens | tokens per run | Keep reasoning separate when the provider reports it separately. |
|
|
275
|
+
| Reasoning tokens | tokens per run | Use the provider's billing rule. Do not assume that they are free. |
|
|
276
|
+
| Billable tools | quantity per run | Name each unit and price source. |
|
|
277
|
+
| Active duration | minutes per run | Use this for variable compute. |
|
|
278
|
+
| Artifacts, storage, and network | provider units | Keep each provider line separate. |
|
|
279
|
+
|
|
280
|
+
Convert an interval to an average month with:
|
|
281
|
+
|
|
282
|
+
```text
|
|
283
|
+
runs per month = 730 / interval_hours
|
|
284
|
+
```
|
|
285
|
+
|
|
286
|
+
For example, one run every 24 hours gives `730 / 24 = 30.4167` scheduled
|
|
287
|
+
starts in an average month. A calendar-month report can use the actual count
|
|
288
|
+
instead. Label which method you use.
|
|
289
|
+
|
|
290
|
+
Then calculate the total attempt count:
|
|
291
|
+
|
|
292
|
+
```text
|
|
293
|
+
attempts per month =
|
|
294
|
+
scheduled starts
|
|
295
|
+
× attempts per scheduled start
|
|
296
|
+
× (1 + delegated runs per attempt)
|
|
297
|
+
```
|
|
298
|
+
|
|
299
|
+
Use the last factor only when every attempt creates that number of delegated
|
|
300
|
+
runs. If delegation varies, list the parent and child workloads separately.
|
|
301
|
+
|
|
302
|
+
## Calculate model cost once
|
|
303
|
+
|
|
304
|
+
Use a price book with a source and an effective date. For a model with separate
|
|
305
|
+
text-token prices:
|
|
306
|
+
|
|
307
|
+
```text
|
|
308
|
+
model cost per run =
|
|
309
|
+
input_tokens × input_price / 1,000,000
|
|
310
|
+
+ cache_read_tokens × cache_read_price / 1,000,000
|
|
311
|
+
+ cache_write_tokens × cache_write_price / 1,000,000
|
|
312
|
+
+ output_tokens × output_price / 1,000,000
|
|
313
|
+
+ provider-defined reasoning cost
|
|
314
|
+
```
|
|
315
|
+
|
|
316
|
+
The same workload profile and price book MUST produce the same model line on
|
|
317
|
+
every execution surface. Do not recalculate model cost inside an Actions or
|
|
318
|
+
Kubernetes estimate.
|
|
319
|
+
|
|
320
|
+
Keep two cost fields when evidence is available:
|
|
321
|
+
|
|
322
|
+
- **Provider-reported cost** is the charge that the provider returned for the
|
|
323
|
+
exchange. Keep its amount and currency unchanged.
|
|
324
|
+
- **Price-book estimate** is a derived amount. Keep the source and effective
|
|
325
|
+
date with it.
|
|
326
|
+
|
|
327
|
+
Do not replace one with the other. Do not present a derived estimate as an
|
|
328
|
+
observed provider charge.
|
|
329
|
+
|
|
330
|
+
## Worked case: one organization report each day
|
|
331
|
+
|
|
332
|
+
The example starts from an observed AI Outfitter organization-report flow. The
|
|
333
|
+
2026-W32 report covered 16 repositories. Its KPI JSON was 21,020 bytes. Its
|
|
334
|
+
rendered Markdown was 11,719 bytes. Scripts collected and rendered the facts.
|
|
335
|
+
The model wrote only a short Highlights paragraph.
|
|
336
|
+
|
|
337
|
+
Those report dimensions are observed. The retained artifacts do not contain a
|
|
338
|
+
token total. Therefore, these token bands are estimated:
|
|
339
|
+
|
|
340
|
+
| Band | Input | Cache read | Output | Total per run |
|
|
341
|
+
| -------- | ------: | ---------: | -----: | ------------: |
|
|
342
|
+
| Compact | 10,000 | 20,000 | 2,000 | 32,000 |
|
|
343
|
+
| Expected | 25,000 | 75,000 | 5,000 | 105,000 |
|
|
344
|
+
| High | 100,000 | 300,000 | 15,000 | 415,000 |
|
|
345
|
+
|
|
346
|
+
All values use the OpenAI half of the [price book](#price-book) above. Keep the
|
|
347
|
+
base estimate and the five-times Luna reserve separate.
|
|
348
|
+
|
|
349
|
+
At `730 / 24 = 30.4167` runs per month, the model estimates are:
|
|
350
|
+
|
|
351
|
+
| Band | Monthly tokens | GPT-5.6 Sol | GPT-5.6 Terra | GPT-5.6 Luna | GPT-5.4 mini |
|
|
352
|
+
| -------- | -------------: | ----------: | ------------: | -----------: | -----------: |
|
|
353
|
+
| Compact | 0.973M | $3.65 | $1.46 | $0.15 | $0.55 |
|
|
354
|
+
| Expected | 3.194M | $9.51 | $3.80 | $0.38 | $1.43 |
|
|
355
|
+
| High | 12.623M | $33.46 | $13.38 | $1.34 | $5.02 |
|
|
356
|
+
|
|
357
|
+
The expected model cost is $0.313 for each run on Sol, $0.125 on Terra,
|
|
358
|
+
$0.013 on Luna, and $0.047 on GPT-5.4 mini. The five-times Luna reserve changes
|
|
359
|
+
its expected monthly planning line from $0.38 to $1.90.
|
|
360
|
+
|
|
361
|
+
These values exclude everything in
|
|
362
|
+
[What this figure excludes](#what-this-figure-excludes), and also tax.
|
|
363
|
+
|
|
364
|
+
This case sits far below the non-stop ceiling at the top of this guide, because
|
|
365
|
+
the model writes one paragraph and scripts do the rest.
|
|
366
|
+
|
|
367
|
+
## Compare execution surfaces
|
|
368
|
+
|
|
369
|
+
Start with the same expected daily-report model line. Then add only the
|
|
370
|
+
surface-specific substrate.
|
|
371
|
+
|
|
372
|
+
| Surface | Fixed substrate | Variable compute |
|
|
373
|
+
| ------------------------------ | --------------------------------------------------------------- | --------------------------------------------------------------- |
|
|
374
|
+
| GitHub-hosted Actions | none for an otherwise idle workload | billable runner minutes after included quotas |
|
|
375
|
+
| Self-hosted Actions | allocated runner and operations cost | per-run cost when your allocation method charges it |
|
|
376
|
+
| Local run | allocated workstation cost, if the organization accounts for it | active process time, if the organization accounts for it |
|
|
377
|
+
| Scheduled Kubernetes Job | none when no workload-specific capacity stays allocated | pod resource time for every attempt |
|
|
378
|
+
| Delegated Kubernetes Job | none when no workload-specific capacity stays allocated | child pod resource time, recorded under its parent run |
|
|
379
|
+
| Resident Kubernetes Deployment | allocated resources for 730 hours | extra per-run resources that are not in the resident allocation |
|
|
380
|
+
|
|
381
|
+
This gives two different substrate equations for the same model line:
|
|
382
|
+
|
|
383
|
+
```text
|
|
384
|
+
Actions monthly cost =
|
|
385
|
+
model and tool cost for all attempts
|
|
386
|
+
+ hosted billable minutes × hosted runner price
|
|
387
|
+
+ self-hosted runner allocation
|
|
388
|
+
+ storage
|
|
389
|
+
+ network
|
|
390
|
+
```
|
|
391
|
+
|
|
392
|
+
```text
|
|
393
|
+
Resident monthly cost =
|
|
394
|
+
resident hourly allocation × 730
|
|
395
|
+
+ model and tool cost for all attempts
|
|
396
|
+
+ storage
|
|
397
|
+
+ network
|
|
398
|
+
```
|
|
399
|
+
|
|
400
|
+
An idle resident has zero model cost when it makes no model exchange. It still
|
|
401
|
+
has its 730-hour substrate allocation. A scheduled Job has no workload-specific
|
|
402
|
+
compute cost while no pod exists, unless the cluster allocation method assigns
|
|
403
|
+
idle capacity to it.
|
|
404
|
+
|
|
405
|
+
See the [Actions billing-inputs research note](https://github.com/ai-outfitter/actions/blob/main/docs/research/inference-pricing.md)
|
|
406
|
+
for the runner-minute form. The
|
|
407
|
+
[Agent Operator](https://github.com/ai-outfitter/agent-operator) resident and
|
|
408
|
+
Job forms are not documented yet.
|
|
409
|
+
|
|
410
|
+
## Retries, failures, and subagents
|
|
411
|
+
|
|
412
|
+
Count every attempt once. A failed attempt can incur model, tool, runner, and
|
|
413
|
+
network costs even when it produces no artifact. A retry MUST use a new attempt
|
|
414
|
+
number. Do not overwrite or hide the first attempt.
|
|
415
|
+
|
|
416
|
+
A delegated run MUST carry its parent run. Add the child cost once to the
|
|
417
|
+
workload total. Do not also copy the child model exchanges into the parent's
|
|
418
|
+
direct model line.
|
|
419
|
+
|
|
420
|
+
If a workflow platform retries a whole job, record that job as a new attempt.
|
|
421
|
+
If a harness retries one provider request inside the same attempt, record each
|
|
422
|
+
model exchange and aggregate them once into that attempt.
|
|
423
|
+
|
|
424
|
+
## Unknown and partial usage
|
|
425
|
+
|
|
426
|
+
Use these states:
|
|
427
|
+
|
|
428
|
+
| State | Meaning |
|
|
429
|
+
| ------------- | -------------------------------------------------------------------------------------------------- |
|
|
430
|
+
| `complete` | The record contains all usage fields that the provider and harness expose. It has no declared gap. |
|
|
431
|
+
| `partial` | The record contains some usage fields and lists each unavailable field. |
|
|
432
|
+
| `unavailable` | The record contains no usable usage counters and lists the gap. |
|
|
433
|
+
|
|
434
|
+
Use `null` for an unavailable counter. Use `0` only when the provider or
|
|
435
|
+
harness observed zero. A monthly view MUST render unavailable cost as unknown.
|
|
436
|
+
It MUST NOT render it as `$0`.
|
|
437
|
+
|
|
438
|
+
Do not compare two totals until their boundaries match. State whether each
|
|
439
|
+
total includes retries, delegated runs, tools, storage, network, quotas, and
|
|
440
|
+
tax.
|
|
441
|
+
|
|
442
|
+
## Replace estimates with measurements
|
|
443
|
+
|
|
444
|
+
Retain one machine-readable record for each run and model exchange. After 30
|
|
445
|
+
measured daily reports, replace the planning bands with the median, 90th
|
|
446
|
+
percentile, and maximum successful-run cost. Report failed attempts and retries
|
|
447
|
+
separately.
|
|
448
|
+
|
|
449
|
+
Only one execution surface can supply that record today:
|
|
450
|
+
|
|
451
|
+
| Surface | Usage record | Status |
|
|
452
|
+
| ----------------------- | ------------------------------------------ | ----------- |
|
|
453
|
+
| Local Claude Code | One `usage` object for each assistant turn | Available |
|
|
454
|
+
| Local `codex` | The rollout file carries no token field | Unavailable |
|
|
455
|
+
| Agent Operator resident | The Pi session file carries no token field | Unavailable |
|
|
456
|
+
| GitHub Actions run | An HTML transcript artifact, with no total | Unavailable |
|
|
457
|
+
|
|
458
|
+
Therefore a comparison across surfaces is not possible yet. Close this gap
|
|
459
|
+
before you replace any estimate above. Two proposals do it:
|
|
460
|
+
[Pensieve pull request #22](https://github.com/ai-outfitter/pensieve/pull/22)
|
|
461
|
+
for the record contract, and
|
|
462
|
+
[Actions pull request #42](https://github.com/ai-outfitter/actions/pull/42)
|
|
463
|
+
for the runner-minute inputs. Neither is merged.
|
|
464
|
+
|
|
465
|
+
This guide defines the method and the record contract. It does not include a
|
|
466
|
+
collector, a calculator command, or a billing dashboard.
|
|
@@ -31,6 +31,10 @@ Runtime and account state is not configuration and stays in `~/.claude` untouche
|
|
|
31
31
|
|
|
32
32
|
This is the same boundary [state persistence](./state.md) enforces at run time: configuration lives in the tree, mutable state lives with the harness.
|
|
33
33
|
|
|
34
|
+
Staying native does not mean being ignored. A Claude run inherits this state by default, so the
|
|
35
|
+
permissions, hooks, plugins, trust, and MCP servers listed above apply to an Outfitter-launched
|
|
36
|
+
session exactly as they do to a native one. `--isolated` is what leaves them behind.
|
|
37
|
+
|
|
34
38
|
## After porting
|
|
35
39
|
|
|
36
40
|
Your resources are now protocol resources. Reference them by slug from an agent's loadout like anything else:
|
|
@@ -23,6 +23,7 @@ In a standalone `.agents` repository the repository root is the tree, so the fil
|
|
|
23
23
|
# .agents/settings.yml
|
|
24
24
|
default_agent: engineer # which agent runs by default
|
|
25
25
|
default_harness: pi # which harness to launch: pi, claude, or codex
|
|
26
|
+
isolation: inherit # inherit (default) or isolated; see below. Honored only from ~/.agents.
|
|
26
27
|
|
|
27
28
|
# Where protocol resources come from, beyond this tree and ~/.agents.
|
|
28
29
|
sources:
|
|
@@ -47,6 +48,7 @@ telemetry:
|
|
|
47
48
|
```
|
|
48
49
|
|
|
49
50
|
- `default_agent` / `default_harness` — which agent plain `outfitter` runs, and the harness it launches in.
|
|
51
|
+
- `isolation` — whether a run stands on the harness configuration already on this machine. `inherit`, the default, layers the composition over it, so a Claude run keeps your workspace trust, permissions, credentials, plugins, and MCP servers. `isolated` launches from the composition alone, which is what a reproducible CI or container run wants; `--isolated` selects it for one run. Only Claude has an inherit path today. This key is honored **only** from your own `~/.agents` settings: a checked-in project or a remote catalog must not decide how much of your machine a profile it ships can see.
|
|
50
52
|
- `sources` — ordered list of remote or local `.agents` payloads. Remote entries (`github:` / `uri:`) accept `ref:` pinning and an optional `path:` to the payload inside the repository; see [Catalogs](./catalogs.md) for conventions and trust guidance.
|
|
51
53
|
- `remote_settings` — shared settings a repository distributes; cached locally and merged below your project and user settings, so anything you set locally wins.
|
|
52
54
|
- `cache_directory` — the repository cache root used consistently by sync, remote settings, remote
|
|
@@ -76,7 +76,7 @@ Undeclared writes governed by `unknown: prompt` cannot be persisted because they
|
|
|
76
76
|
|
|
77
77
|
## Temporary directory cleanup
|
|
78
78
|
|
|
79
|
-
Baked composition directories are created under the system temporary directory and removed automatically when the Outfitter process exits or receives a handled signal. Removal deletes symlink entries without following them, so the durable auth/settings state the links point at is never touched. Pass `--
|
|
79
|
+
Baked composition directories are created under the system temporary directory and removed automatically when the Outfitter process exits or receives a handled signal. Removal deletes symlink entries without following them, so the durable auth/settings state the links point at is never touched. Pass `--retain-projection` to keep the directory for inspection; Outfitter prints its path.
|
|
80
80
|
|
|
81
81
|
Each startup also best-effort sweeps `outfitter-*` directories older than seven days from the temporary root. The sweep never follows symlinks, so a stale directory's links are removed while their targets survive.
|
|
82
82
|
|
|
@@ -190,7 +190,15 @@ The last form is how a resident or in-cluster agent keeps continuity across rest
|
|
|
190
190
|
|
|
191
191
|
## Claude Code state paths
|
|
192
192
|
|
|
193
|
-
|
|
193
|
+
By default a Claude run inherits the machine's own configuration: Outfitter sets no
|
|
194
|
+
`CLAUDE_CONFIG_DIR`, and the composition reaches the session as a plugin directory instead. Claude
|
|
195
|
+
reads and writes `~/.claude` exactly as it does in a native session, so credentials, workspace
|
|
196
|
+
trust, permission approvals, and session history need no bridge at all — there is nothing to seed
|
|
197
|
+
and nothing to copy back, and nothing Outfitter does can race your other Claude sessions. The rest
|
|
198
|
+
of this section describes an **isolated** run (`--isolated`, or `isolation: isolated` in your
|
|
199
|
+
`~/.agents/settings.yml`), where the projection is the whole configuration.
|
|
200
|
+
|
|
201
|
+
Under isolation, Claude credentials need a narrow adapter bridge in addition to the path-keyed state below. Claude
|
|
194
202
|
reads `.credentials.json` and `.claude.json` directly from `CLAUDE_CONFIG_DIR`; the ephemeral
|
|
195
203
|
projection gives `.credentials.json` no durable home, and `.claude.json`'s native location
|
|
196
204
|
(`~/.claude.json`, outside `~/.claude`) does not share its config-dir-relative path. Outfitter
|
|
@@ -203,11 +211,11 @@ instead of copying the projected credentials back. Claude MCP OAuth tokens
|
|
|
203
211
|
live under `mcpOAuth` in `.credentials.json`, keyed by `<serverName>|<hash>`, so server
|
|
204
212
|
authorizations acquired in an Outfitter-launched Claude session persist across runs through that
|
|
205
213
|
whole-file copy-back. Outfitter never copies the full machine-local `~/.claude.json` into a
|
|
206
|
-
projection or merges its other projected state back. Trust accepted inside an
|
|
207
|
-
therefore discarded, so
|
|
208
|
-
natively.
|
|
214
|
+
projection or merges its other projected state back. Trust accepted inside an isolated session is
|
|
215
|
+
therefore discarded, so an isolated run prompts for trust in a workspace that was never trusted
|
|
216
|
+
natively — which is one reason isolation is not the default.
|
|
209
217
|
|
|
210
|
-
Claude session history has a second narrow bridge because `CLAUDE_CONFIG_DIR` also redirects
|
|
218
|
+
Isolated Claude session history has a second narrow bridge because `CLAUDE_CONFIG_DIR` also redirects
|
|
211
219
|
Claude's native `projects/` tree into the temporary projection. Before launch, Outfitter derives
|
|
212
220
|
Claude's project slug from the absolute working directory and copies only that slug directory from
|
|
213
221
|
`~/.claude/projects/`. This keeps other projects' transcripts out of the projection while making
|
|
@@ -217,6 +225,9 @@ the projection's slug directories back into `~/.claude/projects/` atomically wit
|
|
|
217
225
|
Durable files are never deleted. A seed or copy-back failure emits a warning and does not replace
|
|
218
226
|
Claude's exit code or error.
|
|
219
227
|
|
|
228
|
+
These declared paths describe the isolated strategy; an inherited run writes to the native
|
|
229
|
+
locations directly and declares nothing.
|
|
230
|
+
|
|
220
231
|
The Claude Code adapter declares these paths:
|
|
221
232
|
|
|
222
233
|
```yaml
|
|
@@ -46,13 +46,14 @@ Tasks and bake are not in this matrix — they are the subject of a [separate up
|
|
|
46
46
|
|
|
47
47
|
## Claude Code notes
|
|
48
48
|
|
|
49
|
-
- **
|
|
49
|
+
- **Your configuration comes first** — by default a Claude run stands on the configuration already on the machine. Outfitter sets no `CLAUDE_CONFIG_DIR`; it declares the baked composition a Claude plugin and passes it through `--plugin-dir`, so the session keeps your workspace trust, `~/.claude/settings.json` permissions, credentials, plugins, and configured MCP servers, and the profile's skills, subagents, and prompts layer on top. Nothing is seeded and nothing is copied back, because Claude is reading and writing its real configuration directory throughout. Pass `--isolated`, or set `isolation: isolated` in your `~/.agents/settings.yml`, to launch from the composition alone — the reproducible form for CI and containers, and what the remaining bullets in this section describe. If the installed Claude is too old to load a plugin directory, Outfitter falls back to an isolated run and says so rather than failing the launch.
|
|
50
|
+
- **Isolated config and session state** — an isolated run points `CLAUDE_CONFIG_DIR` at the baked composition. Before launch it copies only the current working directory's history from `~/.claude/projects/<project-slug>/` into the projection, so native `--continue` and `--resume` work without exposing other projects. After every successful or failed launch it atomically merges new or changed session files from every projected slug back into `~/.claude/projects/` with mode `0600`, never deleting durable history. Session bridge failures warn without masking the Claude exit. Outfitter also declares Claude state paths (`settings.json`, `agents/`, `skills/`, `commands/`, `plugins/`, `projects/`) for [state persistence](./state.md), and can [symlink a ported `~/.claude`](./porting-claude.md) so native use keeps working. MCP configuration from that port is no longer auto-discovered by Outfitter-launched Claude runs; those servers apply only when an agent selects them by slug. See the next bullet.
|
|
50
51
|
- **Credentials, onboarding, and workspace trust** — before launch, Outfitter copies `~/.claude/.credentials.json` to the temporary root as `.credentials.json` with mode `0600`. The projected `.claude.json` contains `oauthAccount` and `hasCompletedOnboarding` when those keys are present in durable `~/.claude.json`. It also contains `projects[<cwd>].hasTrustDialogAccepted: true` only when that exact accepted trust decision already exists there; other projects and unrelated machine state are not copied. After any successful or failed launch, a `.credentials.json` changed by the run is copied back wholesale and `oauthAccount` is atomically merged into durable `.claude.json` without replacing unrelated keys. If the durable credentials also changed after seeding, Outfitter preserves that concurrent refresh and warns instead of copying back. MCP OAuth tokens live under `mcpOAuth` in `.credentials.json`, keyed by `<serverName>|<hash>`, so authorizations acquired in an Outfitter-launched Claude session persist across runs. Other projected `.claude.json` state, including trust accepted during the session, is discarded; a workspace that has never been trusted by native Claude therefore prompts again on every run.
|
|
51
|
-
- **MCP servers** — every Claude launch passes the generated `mcp.json` through `--mcp-config
|
|
52
|
-
- **Subagents** — selected `agents/<id>` definitions are materialized into
|
|
52
|
+
- **MCP servers** — every Claude launch passes the generated `mcp.json` through `--mcp-config`. An inherited run stops there, so the composition's servers merge with the ones already configured on the machine: selecting a server says what the profile needs, not what the user may not have. An isolated run adds `--strict-mcp-config`, which excludes MCP servers from user or project configuration, `.claude.json`, and plugins so only the composition's servers are active.
|
|
53
|
+
- **Subagents** — selected `agents/<id>` definitions are materialized into the composition's agents directory. An inherited run loads them under the plugin's name (`<profile>:<subagent>`); an isolated run finds them natively under `CLAUDE_CONFIG_DIR`.
|
|
53
54
|
- **Skills (Partial)** — selected skills are materialized into the config directory's skills surface; remaining gaps are tracked per release. The bundled Outfitter skill ships through the plugin channel.
|
|
54
55
|
- **Model selection (Partial)** — model maps to `--model` and thinking level to `--effort`; provider selection is not projected for Claude and warns if requested.
|
|
55
|
-
- **Hooks
|
|
56
|
+
- **Hooks** — Outfitter does not project hook configuration for Claude, and there is no portable protocol hooks resource yet. An inherited run keeps the hooks in your own `~/.claude/settings.json`; an isolated run has none. See [Hooks](./hooks.md).
|
|
56
57
|
- **Tool availability** — `tools.allow` (after `tools.deny` removes entries) maps to both `--tools` (_availability_: an unlisted builtin is not in the session) and `--allowedTools` (_permission_: the granted tools are pre-approved, so a headless session is not stopped by a prompt); `tools.deny` always maps to `--disallowedTools`, including when both are declared, and a bare denied name removes the tool from context per Claude's docs. An allowlist that `tools.deny` empties maps to `--tools ""`, Claude's documented "disable all tools" form. Caveat: per the CLI reference, `--tools` governs the built-in set only — MCP tools (`mcp__server__*`) are unaffected and are governed by which MCP servers the loadout selects, so `--tools ""` is not exactly pi's zero-tool session when MCP servers are present. Claude's behavior here comes from `claude --help` and the CLI reference, not local measurement.
|
|
57
58
|
- **DeepWork jobs** — job selection is Pi-only today and warns on Claude.
|
|
58
59
|
- **Bundled Outfitter skill** — every launch also publishes Outfitter's own self-documentation skill as a bundled plugin, so the agent can explain Outfitter and this launch's configuration.
|
package/package.json
CHANGED
|
@@ -5,16 +5,16 @@
|
|
|
5
5
|
"type": "object",
|
|
6
6
|
"required": ["name"],
|
|
7
7
|
"properties": {
|
|
8
|
-
"name": { "
|
|
8
|
+
"name": { "$ref": "#/$defs/agentSlug" },
|
|
9
9
|
"label": { "type": "string", "minLength": 1 },
|
|
10
10
|
"description": { "type": "string" },
|
|
11
11
|
"inherits": {
|
|
12
12
|
"oneOf": [
|
|
13
|
-
{ "
|
|
13
|
+
{ "$ref": "#/$defs/agentSlug" },
|
|
14
14
|
{
|
|
15
15
|
"type": "array",
|
|
16
16
|
"minItems": 1,
|
|
17
|
-
"items": { "
|
|
17
|
+
"items": { "$ref": "#/$defs/agentSlug" }
|
|
18
18
|
}
|
|
19
19
|
],
|
|
20
20
|
"description": "Parent agent slug or ordered parent slug list composed parent-first before this agent."
|
|
@@ -36,7 +36,7 @@
|
|
|
36
36
|
"description": "Skill slugs resolved from agents/<name>/skills first, then catalog-wide skills."
|
|
37
37
|
},
|
|
38
38
|
"subagents": {
|
|
39
|
-
"$ref": "#/$defs/
|
|
39
|
+
"$ref": "#/$defs/agentSlugList",
|
|
40
40
|
"description": "Agent slugs delegated to. Resolved catalog-wide (a delegate is a shared agent, not nested under agents/<name>)."
|
|
41
41
|
},
|
|
42
42
|
"mcp": {
|
|
@@ -64,6 +64,15 @@
|
|
|
64
64
|
},
|
|
65
65
|
"additionalProperties": true,
|
|
66
66
|
"$defs": {
|
|
67
|
+
"agentSlug": {
|
|
68
|
+
"type": "string",
|
|
69
|
+
"pattern": "^[a-z0-9]+(?:-[a-z0-9]+)*(?:\\.[a-z0-9]+(?:-[a-z0-9]+)*)*$",
|
|
70
|
+
"maxLength": 64
|
|
71
|
+
},
|
|
72
|
+
"agentSlugList": {
|
|
73
|
+
"type": "array",
|
|
74
|
+
"items": { "$ref": "#/$defs/agentSlug" }
|
|
75
|
+
},
|
|
67
76
|
"toolName": {
|
|
68
77
|
"type": "string",
|
|
69
78
|
"minLength": 1,
|
|
@@ -6,6 +6,10 @@
|
|
|
6
6
|
"properties": {
|
|
7
7
|
"default_agent": { "type": "string", "minLength": 1 },
|
|
8
8
|
"default_harness": { "enum": ["pi", "claude", "codex"] },
|
|
9
|
+
"isolation": {
|
|
10
|
+
"enum": ["inherit", "isolated"],
|
|
11
|
+
"description": "Whether a run stands on the machine's native harness configuration (inherit, the default) or on the projection alone (isolated). Honored only from home-scope settings."
|
|
12
|
+
},
|
|
9
13
|
"cache_directory": { "type": "string", "minLength": 1 },
|
|
10
14
|
"state_persistence": {
|
|
11
15
|
"type": "object",
|