@ai-outfitter/outfitter 1.8.1 → 1.10.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (34) hide show
  1. package/README.md +1 -1
  2. package/dist/agents/ClaudeConfigStrategy.d.ts +15 -0
  3. package/dist/agents/ClaudeConfigStrategy.js +53 -0
  4. package/dist/agents/ClaudeConfigStrategy.js.map +1 -0
  5. package/dist/cli/commands/RunAgentCommand.d.ts +5 -0
  6. package/dist/cli/commands/RunAgentCommand.js +30 -8
  7. package/dist/cli/commands/RunAgentCommand.js.map +1 -1
  8. package/dist/projection/Materialize.d.ts +7 -0
  9. package/dist/projection/Materialize.js +24 -0
  10. package/dist/projection/Materialize.js.map +1 -1
  11. package/dist/projection/ProjectHarness.js +48 -10
  12. package/dist/projection/ProjectHarness.js.map +1 -1
  13. package/dist/projection/Projection.d.ts +9 -1
  14. package/dist/schemas/agent.schema.json +13 -4
  15. package/dist/schemas/settings.schema.json +4 -0
  16. package/dist/settings/Settings.d.ts +14 -0
  17. package/dist/settings/Settings.js +7 -0
  18. package/dist/settings/Settings.js.map +1 -1
  19. package/dist/settings/SettingsLoader.js +4 -2
  20. package/dist/settings/SettingsLoader.js.map +1 -1
  21. package/dist/settings/SettingsMerger.js +3 -0
  22. package/dist/settings/SettingsMerger.js.map +1 -1
  23. package/dist/setup/Setup.js +7 -4
  24. package/dist/setup/Setup.js.map +1 -1
  25. package/docs/documentation/README.md +2 -0
  26. package/docs/documentation/cli.md +8 -6
  27. package/docs/documentation/cost-estimation.md +466 -0
  28. package/docs/documentation/porting-claude.md +4 -0
  29. package/docs/documentation/settings.md +2 -0
  30. package/docs/documentation/state.md +17 -6
  31. package/docs/documentation/support-matrix.md +5 -4
  32. package/package.json +1 -1
  33. package/src/schemas/agent.schema.json +13 -4
  34. package/src/schemas/settings.schema.json +4 -0
@@ -0,0 +1,466 @@
1
+ # Estimate agent workload cost
2
+
3
+ This table estimates what one agent that never stops costs in model tokens.
4
+ Every figure comes from a dated price book and an estimated token rate, not
5
+ from an invoice. The day column is 24 hours at the stated rate. The month
6
+ column is 730 hours, the average month.
7
+
8
+ | Provider | Model | Steady: day / month | Saturated: day / month |
9
+ | --------- | ---------------- | ------------------: | ---------------------: |
10
+ | Anthropic | Claude Fable 5 | $182.40 / $5,548 | $408.00 / $12,410 |
11
+ | OpenAI | GPT-5.6 Sol | $98.40 / $2,993 | $216.00 / $6,570 |
12
+ | Anthropic | Claude Opus 5 | $91.20 / $2,774 | $204.00 / $6,205 |
13
+ | OpenAI | GPT-5.6 Terra | $39.36 / $1,197 | $86.40 / $2,628 |
14
+ | Anthropic | Claude Sonnet 5 | $36.48 / $1,110 | $81.60 / $2,482 |
15
+ | Anthropic | Claude Haiku 4.5 | $18.24 / $555 | $40.80 / $1,241 |
16
+ | OpenAI | GPT-5.4 mini | $14.76 / $449 | $32.40 / $986 |
17
+ | OpenAI | GPT-5.6 Luna | $3.94 / $120 | $8.64 / $263 |
18
+
19
+ Four notes apply to every row:
20
+
21
+ - **The two rates.** Steady is ordinary agent turns with repeated instructions:
22
+ 400K input, 600K cache-read, and 60K output tokens for each active hour.
23
+ Saturated is continuous agentic coding with large tool results:
24
+ 1M / 2M / 100K.
25
+ - **Scope.** This is the model line only. It excludes substrate, tools,
26
+ subagents, and cache writes. See
27
+ [What this figure excludes](#what-this-figure-excludes).
28
+ - **Luna reserve.** Apply a five-times reserve to GPT-5.6 Luna until an invoice
29
+ settles its disputed price. That gives $19.68 / $599 steady and
30
+ $43.20 / $1,314 saturated.
31
+ - **Part-time agents.** Cost scales with active hours. Eight hours on each
32
+ working day is about 176 hours, which is 24 percent of the month column.
33
+
34
+ At the same token rate, GPT-5.6 Sol costs about 25 times GPT-5.6 Luna and
35
+ Claude Fable 5 costs about 46 times. Apply the Luna reserve and those spreads
36
+ fall to 5 and 9 times. Against Claude Haiku 4.5, the cheapest model with an
37
+ undisputed price, they are 5.4 and 10 times. The two rates change the cost of
38
+ one model by about 2.2 times. Model choice and work intensity are separate
39
+ controls.
40
+
41
+ This guide calculates a planning estimate. It does not read invoices or collect
42
+ usage. See
43
+ [Replace estimates with measurements](#replace-estimates-with-measurements) for
44
+ the record that can replace each estimate later.
45
+
46
+ ## Price book
47
+
48
+ Two providers, two capture dates. Prices are United States dollars for one
49
+ million text tokens.
50
+
51
+ OpenAI standard text-token prices, captured 2026-08-15 from the
52
+ [OpenAI model documentation](https://developers.openai.com/api/docs/models):
53
+
54
+ | Model | Documented role | Input | Cache read | Output |
55
+ | ------------- | ----------------------------------------------- | ----: | ---------: | -----: |
56
+ | GPT-5.6 Sol | Frontier tier | $5.00 | $0.50 | $30.00 |
57
+ | GPT-5.6 Terra | Balance of intelligence and cost | $2.00 | $0.20 | $12.00 |
58
+ | GPT-5.4 mini | High-volume coding, computer use, and subagents | $0.75 | $0.075 | $4.50 |
59
+ | GPT-5.6 Luna | Cost-sensitive, high-volume tier | $0.20 | $0.02 | $1.20 |
60
+
61
+ Anthropic standard text-token prices, captured 2026-08-17 from the
62
+ [Claude platform pricing page](https://platform.claude.com/docs/en/about-claude/pricing):
63
+
64
+ | Model | Documented role | Input | Cache read | Output |
65
+ | ---------------- | ------------------------------------------ | -----: | ---------: | -----: |
66
+ | Claude Fable 5 | Demanding reasoning and long-horizon work | $10.00 | $1.00 | $50.00 |
67
+ | Claude Opus 5 | Complex agentic coding and enterprise work | $5.00 | $0.50 | $25.00 |
68
+ | Claude Sonnet 5 | Balance of speed and intelligence | $2.00 | $0.20 | $10.00 |
69
+ | Claude Haiku 4.5 | Fastest and most cost-effective | $1.00 | $0.10 | $5.00 |
70
+
71
+ Three properties of these tables control every figure in this guide:
72
+
73
+ - The Anthropic price vectors are exactly proportional on every token class.
74
+ `Fable 5 : Opus 5 : Sonnet 5 : Haiku 4.5 = 10 : 5 : 2 : 1`. Therefore any
75
+ Anthropic figure in this guide scales by that ratio, whatever the token mix
76
+ is. Use this property to check the arithmetic.
77
+ - **Neither table contains a cache-write price, and both providers charge
78
+ one.** An OpenAI explicit cache write costs 1.25 times the uncached-input
79
+ price. An Anthropic cache write costs 1.25 times input for a 5-minute
80
+ lifetime and 2 times input for a 1-hour lifetime. Every figure in this guide
81
+ omits that line, so every figure understates the cost. See
82
+ [Correction from measured usage](#correction-from-measured-usage) for the
83
+ size of that error.
84
+ - **The GPT-5.6 Luna price is disputed.** The 2026-08-15 capture and a later
85
+ live re-check of the model catalog, the comparison page, and the Luna page
86
+ all gave $0.20 / $0.02 / $1.20. Official-domain search results crawled two
87
+ weeks earlier gave $1.00 / $0.10 / $6.00, which is five times more. No
88
+ invoice has settled it. That ratio is the five-times reserve this guide
89
+ applies to every Luna figure.
90
+
91
+ In the 2026-08-17 capture, the Claude Sonnet 5 introductory price of
92
+ $2.00 / $10.00 is the standard price, and a previously scheduled increase to
93
+ $3.00 / $15.00 is no longer listed. An estimate built on the increase is about
94
+ 50 percent too high.
95
+
96
+ ## How this figure is calculated
97
+
98
+ Three inputs produce every number in the table above: a dated price book, an
99
+ estimated token rate, and the active hours in the period.
100
+
101
+ ### Step 1 — select a token rate
102
+
103
+ The rate is the only estimated input. Two rates bound realistic agent work.
104
+ Both come from the workload profile, not from a measured ledger:
105
+
106
+ | Rate | Per active hour: input / cache read / output | Derivation |
107
+ | --------- | -------------------------------------------- | ---------------------------------------------------------------- |
108
+ | Steady | 400K / 600K / 60K | 100K / 150K / 15K for each run, 15 minutes each run, 4 each hour |
109
+ | Saturated | 1M / 2M / 100K | 500K / 1M / 50K for each run, 30 minutes each run, 2 each hour |
110
+
111
+ No observed provider record supplies these two rates. Every dollar figure in
112
+ this guide inherits that uncertainty.
113
+
114
+ ### Step 2 — calculate the cost of one active agent-hour
115
+
116
+ Apply the price book to the selected rate. For Claude Opus 5 at the steady
117
+ rate:
118
+
119
+ ```text
120
+ 400,000 input × $5.00 / 1,000,000 = $2.00
121
+ + 600,000 cache read × $0.50 / 1,000,000 = $0.30
122
+ + 60,000 output × $25.00 / 1,000,000 = $1.50
123
+ -----
124
+ $3.80 per active agent-hour
125
+ ```
126
+
127
+ The general form, with cache writes and reasoning, is in
128
+ [Calculate model cost once](#calculate-model-cost-once). The same calculation
129
+ for every model:
130
+
131
+ | Model | Steady | Saturated |
132
+ | ---------------- | -----: | --------: |
133
+ | Claude Fable 5 | $7.60 | $17.00 |
134
+ | GPT-5.6 Sol | $4.10 | $9.00 |
135
+ | Claude Opus 5 | $3.80 | $8.50 |
136
+ | GPT-5.6 Terra | $1.64 | $3.60 |
137
+ | Claude Sonnet 5 | $1.52 | $3.40 |
138
+ | Claude Haiku 4.5 | $0.76 | $1.70 |
139
+ | GPT-5.4 mini | $0.615 | $1.35 |
140
+ | GPT-5.6 Luna | $0.164 | $0.36 |
141
+
142
+ ### Step 3 — multiply by the active hours in the period
143
+
144
+ A non-stop agent is active for 24 hours on each day and 730 hours in each
145
+ month. For Claude Opus 5 at the steady rate:
146
+
147
+ ```text
148
+ $3.80 × 24 = $91.20 for one day
149
+ $3.80 × 730 = $2,774 for one month
150
+ ```
151
+
152
+ The table at the top of this guide applies these two multiplications to every
153
+ model.
154
+
155
+ ## What this figure excludes
156
+
157
+ The table at the top of this guide is the model line only. The full equation
158
+ has five more terms. The prices quoted below come from the same 2026-08-17
159
+ Anthropic capture as the price book.
160
+
161
+ - **Cache writes.** See the price book above. This is the one excluded term
162
+ with a measured size. See
163
+ [Correction from measured usage](#correction-from-measured-usage).
164
+ - **Substrate.** A non-stop agent holds compute for all 730 hours. Add the
165
+ hourly price of the node or pod. Claude Managed Agents prices this line
166
+ directly at $0.08 for each session-hour, which is $58.40 for a non-stop
167
+ month.
168
+ - **Tool fees.** Anthropic web search costs $10 for 1,000 searches. An agent
169
+ that searches once each minute non-stop adds $438 each month.
170
+ - **Subagents.** A coordinator that runs subagents multiplies the token rate by
171
+ its concurrency. These rates describe one serial thread.
172
+ - **Long-context and residency uplifts.** A GPT-5.6 request above 272,000 input
173
+ tokens can cost more. Anthropic `inference_geo: "us"` adds 1.1 times.
174
+
175
+ ## Correction from measured usage
176
+
177
+ One measured record exists, and it corrects the token mix that the estimate
178
+ assumes.
179
+
180
+ A single workstation retained 2,804 Claude Code session files covering 42
181
+ active days. Each assistant turn carries a `usage` object. Aggregating all
182
+ 221,609 turns gives 45.84 billion tokens with this composition:
183
+
184
+ | Token class | Assumed by the rates | Measured |
185
+ | -------------- | -------------------: | -------: |
186
+ | Uncached input | ~38% | 0.04% |
187
+ | Cache read | ~57% | 96.76% |
188
+ | Cache write | 0%, omitted | 2.82% |
189
+ | Output | ~6% | 0.39% |
190
+
191
+ Two corrections follow:
192
+
193
+ 1. **Uncached input is a rounding error.** Real agent work re-reads a cached
194
+ prefix on almost every turn. A cost model weighted toward uncached input
195
+ prices the wrong thing. Priced on Claude Opus 5, the measured mix costs
196
+ about 4.7 times less for each token than the steady rate does.
197
+ 2. **Cache writes are the second-largest cost line.** The 125,394 Claude Opus 5
198
+ turns in this record used 24.51B cache-read, 703M cache-write, 86M output,
199
+ and 2.08M uncached-input tokens. Priced on Claude Opus 5 and treating every
200
+ cache write as a 5-minute write, that is $12,255 cache read, $4,397 cache
201
+ write, $2,151 output, and $10 uncached input: 65, 23, 11, and 0.05 percent.
202
+ A table that omits a line of that size reports about 77 percent of the model
203
+ cost.
204
+
205
+ Correction 2 is worth about 1.3 times, and correction 1 about 4.7 times. Both
206
+ are smaller than the 25-times spread between model tiers, so model choice
207
+ remains the larger control. Neither correction transfers directly to the tables
208
+ above: the two rates assume no cache writes at all, so their cache-write error
209
+ is unknown rather than 23 percent.
210
+
211
+ This record measures one human driving agents interactively between 2026-07-05
212
+ and 2026-08-17. It does not measure a resident Deployment, and it is not a
213
+ provider invoice. Treat it as evidence about token composition, not as a price
214
+ for the non-stop case.
215
+
216
+ ## Evidence class for each number
217
+
218
+ | Value | Class | Basis |
219
+ | --------------------------- | ---------- | --------------------------------------------------------------- |
220
+ | Model prices | Price book | Dated public captures, 2026-08-15 and 2026-08-17 |
221
+ | 730 hours in a month | Convention | The [`730 / interval_hours` rule](#define-one-workload-profile) |
222
+ | The two token rates | Estimated | Workload profile assumptions, no ledger |
223
+ | Cost per hour and per month | Derived | Price book times the estimated rate |
224
+ | Token-class composition | Measured | 221,609 assistant turns over 42 days |
225
+ | GPT-5.6 Luna price | Disputed | $0.20 / $0.02 / $1.20 against $1.00 / $0.10 / $6.00, no invoice |
226
+ | Substrate and tool prices | Price book | The 2026-08-17 Anthropic capture |
227
+ | Every other excluded term | Unknown | Not zero |
228
+
229
+ A price-book calculation is not an observed provider charge. An unknown value
230
+ is not zero.
231
+
232
+ ## Cost boundary
233
+
234
+ Use one workload profile for every execution surface. Change only the substrate
235
+ inputs when you compare a local run, GitHub Actions, a scheduled Kubernetes
236
+ Job, or a resident Deployment. Use this equation:
237
+
238
+ ```text
239
+ monthly cost =
240
+ fixed substrate cost
241
+ + runs × (model + tools + variable compute)
242
+ + storage
243
+ + network
244
+ ```
245
+
246
+ The terms mean:
247
+
248
+ | Term | Include |
249
+ | ---------------- | ------------------------------------------------------------------------------------- |
250
+ | Fixed substrate | Compute that accrues while no run is active. A resident Deployment is the usual case. |
251
+ | Runs | Every initial attempt, retry, and delegated run in the month. |
252
+ | Model | Input, cache-read, cache-write, output, and reasoning charges for one run. |
253
+ | Tools | Search, browser, image, sandbox, or other billable tool use for one run. |
254
+ | Variable compute | Runner minutes or Job resources that accrue only while a run is active. |
255
+ | Storage | Session data, artifacts, logs, volumes, and backups. |
256
+ | Network | Egress, gateways, and other billed transfer. |
257
+
258
+ An included quota changes the billed amount. It does not change usage. Record
259
+ the usage before you apply the quota.
260
+
261
+ ## Define one workload profile
262
+
263
+ Enter these values once:
264
+
265
+ | Input | Unit | Rule |
266
+ | ------------------------------- | ---------------- | ------------------------------------------------------------------ |
267
+ | Workload identity | stable text | Use the same identity on every surface. |
268
+ | Interval | hours | Use the time between scheduled starts. |
269
+ | Attempts per scheduled start | count | Include the first attempt and expected retries. |
270
+ | Delegated runs per attempt | count | Count each subagent or delegated Job separately. |
271
+ | Input tokens | tokens per run | Keep uncached input separate from cache reads. |
272
+ | Cache-read tokens | tokens per run | Use `null`, not zero, when the provider does not report them. |
273
+ | Cache-write tokens | tokens per run | Use `null`, not zero, when the provider does not report them. |
274
+ | Output tokens | tokens per run | Keep reasoning separate when the provider reports it separately. |
275
+ | Reasoning tokens | tokens per run | Use the provider's billing rule. Do not assume that they are free. |
276
+ | Billable tools | quantity per run | Name each unit and price source. |
277
+ | Active duration | minutes per run | Use this for variable compute. |
278
+ | Artifacts, storage, and network | provider units | Keep each provider line separate. |
279
+
280
+ Convert an interval to an average month with:
281
+
282
+ ```text
283
+ runs per month = 730 / interval_hours
284
+ ```
285
+
286
+ For example, one run every 24 hours gives `730 / 24 = 30.4167` scheduled
287
+ starts in an average month. A calendar-month report can use the actual count
288
+ instead. Label which method you use.
289
+
290
+ Then calculate the total attempt count:
291
+
292
+ ```text
293
+ attempts per month =
294
+ scheduled starts
295
+ × attempts per scheduled start
296
+ × (1 + delegated runs per attempt)
297
+ ```
298
+
299
+ Use the last factor only when every attempt creates that number of delegated
300
+ runs. If delegation varies, list the parent and child workloads separately.
301
+
302
+ ## Calculate model cost once
303
+
304
+ Use a price book with a source and an effective date. For a model with separate
305
+ text-token prices:
306
+
307
+ ```text
308
+ model cost per run =
309
+ input_tokens × input_price / 1,000,000
310
+ + cache_read_tokens × cache_read_price / 1,000,000
311
+ + cache_write_tokens × cache_write_price / 1,000,000
312
+ + output_tokens × output_price / 1,000,000
313
+ + provider-defined reasoning cost
314
+ ```
315
+
316
+ The same workload profile and price book MUST produce the same model line on
317
+ every execution surface. Do not recalculate model cost inside an Actions or
318
+ Kubernetes estimate.
319
+
320
+ Keep two cost fields when evidence is available:
321
+
322
+ - **Provider-reported cost** is the charge that the provider returned for the
323
+ exchange. Keep its amount and currency unchanged.
324
+ - **Price-book estimate** is a derived amount. Keep the source and effective
325
+ date with it.
326
+
327
+ Do not replace one with the other. Do not present a derived estimate as an
328
+ observed provider charge.
329
+
330
+ ## Worked case: one organization report each day
331
+
332
+ The example starts from an observed AI Outfitter organization-report flow. The
333
+ 2026-W32 report covered 16 repositories. Its KPI JSON was 21,020 bytes. Its
334
+ rendered Markdown was 11,719 bytes. Scripts collected and rendered the facts.
335
+ The model wrote only a short Highlights paragraph.
336
+
337
+ Those report dimensions are observed. The retained artifacts do not contain a
338
+ token total. Therefore, these token bands are estimated:
339
+
340
+ | Band | Input | Cache read | Output | Total per run |
341
+ | -------- | ------: | ---------: | -----: | ------------: |
342
+ | Compact | 10,000 | 20,000 | 2,000 | 32,000 |
343
+ | Expected | 25,000 | 75,000 | 5,000 | 105,000 |
344
+ | High | 100,000 | 300,000 | 15,000 | 415,000 |
345
+
346
+ All values use the OpenAI half of the [price book](#price-book) above. Keep the
347
+ base estimate and the five-times Luna reserve separate.
348
+
349
+ At `730 / 24 = 30.4167` runs per month, the model estimates are:
350
+
351
+ | Band | Monthly tokens | GPT-5.6 Sol | GPT-5.6 Terra | GPT-5.6 Luna | GPT-5.4 mini |
352
+ | -------- | -------------: | ----------: | ------------: | -----------: | -----------: |
353
+ | Compact | 0.973M | $3.65 | $1.46 | $0.15 | $0.55 |
354
+ | Expected | 3.194M | $9.51 | $3.80 | $0.38 | $1.43 |
355
+ | High | 12.623M | $33.46 | $13.38 | $1.34 | $5.02 |
356
+
357
+ The expected model cost is $0.313 for each run on Sol, $0.125 on Terra,
358
+ $0.013 on Luna, and $0.047 on GPT-5.4 mini. The five-times Luna reserve changes
359
+ its expected monthly planning line from $0.38 to $1.90.
360
+
361
+ These values exclude everything in
362
+ [What this figure excludes](#what-this-figure-excludes), and also tax.
363
+
364
+ This case sits far below the non-stop ceiling at the top of this guide, because
365
+ the model writes one paragraph and scripts do the rest.
366
+
367
+ ## Compare execution surfaces
368
+
369
+ Start with the same expected daily-report model line. Then add only the
370
+ surface-specific substrate.
371
+
372
+ | Surface | Fixed substrate | Variable compute |
373
+ | ------------------------------ | --------------------------------------------------------------- | --------------------------------------------------------------- |
374
+ | GitHub-hosted Actions | none for an otherwise idle workload | billable runner minutes after included quotas |
375
+ | Self-hosted Actions | allocated runner and operations cost | per-run cost when your allocation method charges it |
376
+ | Local run | allocated workstation cost, if the organization accounts for it | active process time, if the organization accounts for it |
377
+ | Scheduled Kubernetes Job | none when no workload-specific capacity stays allocated | pod resource time for every attempt |
378
+ | Delegated Kubernetes Job | none when no workload-specific capacity stays allocated | child pod resource time, recorded under its parent run |
379
+ | Resident Kubernetes Deployment | allocated resources for 730 hours | extra per-run resources that are not in the resident allocation |
380
+
381
+ This gives two different substrate equations for the same model line:
382
+
383
+ ```text
384
+ Actions monthly cost =
385
+ model and tool cost for all attempts
386
+ + hosted billable minutes × hosted runner price
387
+ + self-hosted runner allocation
388
+ + storage
389
+ + network
390
+ ```
391
+
392
+ ```text
393
+ Resident monthly cost =
394
+ resident hourly allocation × 730
395
+ + model and tool cost for all attempts
396
+ + storage
397
+ + network
398
+ ```
399
+
400
+ An idle resident has zero model cost when it makes no model exchange. It still
401
+ has its 730-hour substrate allocation. A scheduled Job has no workload-specific
402
+ compute cost while no pod exists, unless the cluster allocation method assigns
403
+ idle capacity to it.
404
+
405
+ See the [Actions billing-inputs research note](https://github.com/ai-outfitter/actions/blob/main/docs/research/inference-pricing.md)
406
+ for the runner-minute form. The
407
+ [Agent Operator](https://github.com/ai-outfitter/agent-operator) resident and
408
+ Job forms are not documented yet.
409
+
410
+ ## Retries, failures, and subagents
411
+
412
+ Count every attempt once. A failed attempt can incur model, tool, runner, and
413
+ network costs even when it produces no artifact. A retry MUST use a new attempt
414
+ number. Do not overwrite or hide the first attempt.
415
+
416
+ A delegated run MUST carry its parent run. Add the child cost once to the
417
+ workload total. Do not also copy the child model exchanges into the parent's
418
+ direct model line.
419
+
420
+ If a workflow platform retries a whole job, record that job as a new attempt.
421
+ If a harness retries one provider request inside the same attempt, record each
422
+ model exchange and aggregate them once into that attempt.
423
+
424
+ ## Unknown and partial usage
425
+
426
+ Use these states:
427
+
428
+ | State | Meaning |
429
+ | ------------- | -------------------------------------------------------------------------------------------------- |
430
+ | `complete` | The record contains all usage fields that the provider and harness expose. It has no declared gap. |
431
+ | `partial` | The record contains some usage fields and lists each unavailable field. |
432
+ | `unavailable` | The record contains no usable usage counters and lists the gap. |
433
+
434
+ Use `null` for an unavailable counter. Use `0` only when the provider or
435
+ harness observed zero. A monthly view MUST render unavailable cost as unknown.
436
+ It MUST NOT render it as `$0`.
437
+
438
+ Do not compare two totals until their boundaries match. State whether each
439
+ total includes retries, delegated runs, tools, storage, network, quotas, and
440
+ tax.
441
+
442
+ ## Replace estimates with measurements
443
+
444
+ Retain one machine-readable record for each run and model exchange. After 30
445
+ measured daily reports, replace the planning bands with the median, 90th
446
+ percentile, and maximum successful-run cost. Report failed attempts and retries
447
+ separately.
448
+
449
+ Only one execution surface can supply that record today:
450
+
451
+ | Surface | Usage record | Status |
452
+ | ----------------------- | ------------------------------------------ | ----------- |
453
+ | Local Claude Code | One `usage` object for each assistant turn | Available |
454
+ | Local `codex` | The rollout file carries no token field | Unavailable |
455
+ | Agent Operator resident | The Pi session file carries no token field | Unavailable |
456
+ | GitHub Actions run | An HTML transcript artifact, with no total | Unavailable |
457
+
458
+ Therefore a comparison across surfaces is not possible yet. Close this gap
459
+ before you replace any estimate above. Two proposals do it:
460
+ [Pensieve pull request #22](https://github.com/ai-outfitter/pensieve/pull/22)
461
+ for the record contract, and
462
+ [Actions pull request #42](https://github.com/ai-outfitter/actions/pull/42)
463
+ for the runner-minute inputs. Neither is merged.
464
+
465
+ This guide defines the method and the record contract. It does not include a
466
+ collector, a calculator command, or a billing dashboard.
@@ -31,6 +31,10 @@ Runtime and account state is not configuration and stays in `~/.claude` untouche
31
31
 
32
32
  This is the same boundary [state persistence](./state.md) enforces at run time: configuration lives in the tree, mutable state lives with the harness.
33
33
 
34
+ Staying native does not mean being ignored. A Claude run inherits this state by default, so the
35
+ permissions, hooks, plugins, trust, and MCP servers listed above apply to an Outfitter-launched
36
+ session exactly as they do to a native one. `--isolated` is what leaves them behind.
37
+
34
38
  ## After porting
35
39
 
36
40
  Your resources are now protocol resources. Reference them by slug from an agent's loadout like anything else:
@@ -23,6 +23,7 @@ In a standalone `.agents` repository the repository root is the tree, so the fil
23
23
  # .agents/settings.yml
24
24
  default_agent: engineer # which agent runs by default
25
25
  default_harness: pi # which harness to launch: pi, claude, or codex
26
+ isolation: inherit # inherit (default) or isolated; see below. Honored only from ~/.agents.
26
27
 
27
28
  # Where protocol resources come from, beyond this tree and ~/.agents.
28
29
  sources:
@@ -47,6 +48,7 @@ telemetry:
47
48
  ```
48
49
 
49
50
  - `default_agent` / `default_harness` — which agent plain `outfitter` runs, and the harness it launches in.
51
+ - `isolation` — whether a run stands on the harness configuration already on this machine. `inherit`, the default, layers the composition over it, so a Claude run keeps your workspace trust, permissions, credentials, plugins, and MCP servers. `isolated` launches from the composition alone, which is what a reproducible CI or container run wants; `--isolated` selects it for one run. Only Claude has an inherit path today. This key is honored **only** from your own `~/.agents` settings: a checked-in project or a remote catalog must not decide how much of your machine a profile it ships can see.
50
52
  - `sources` — ordered list of remote or local `.agents` payloads. Remote entries (`github:` / `uri:`) accept `ref:` pinning and an optional `path:` to the payload inside the repository; see [Catalogs](./catalogs.md) for conventions and trust guidance.
51
53
  - `remote_settings` — shared settings a repository distributes; cached locally and merged below your project and user settings, so anything you set locally wins.
52
54
  - `cache_directory` — the repository cache root used consistently by sync, remote settings, remote
@@ -76,7 +76,7 @@ Undeclared writes governed by `unknown: prompt` cannot be persisted because they
76
76
 
77
77
  ## Temporary directory cleanup
78
78
 
79
- Baked composition directories are created under the system temporary directory and removed automatically when the Outfitter process exits or receives a handled signal. Removal deletes symlink entries without following them, so the durable auth/settings state the links point at is never touched. Pass `--debug` to keep the directory for inspection; Outfitter prints its path.
79
+ Baked composition directories are created under the system temporary directory and removed automatically when the Outfitter process exits or receives a handled signal. Removal deletes symlink entries without following them, so the durable auth/settings state the links point at is never touched. Pass `--retain-projection` to keep the directory for inspection; Outfitter prints its path.
80
80
 
81
81
  Each startup also best-effort sweeps `outfitter-*` directories older than seven days from the temporary root. The sweep never follows symlinks, so a stale directory's links are removed while their targets survive.
82
82
 
@@ -190,7 +190,15 @@ The last form is how a resident or in-cluster agent keeps continuity across rest
190
190
 
191
191
  ## Claude Code state paths
192
192
 
193
- Claude credentials need a narrow adapter bridge in addition to the path-keyed state below. Claude
193
+ By default a Claude run inherits the machine's own configuration: Outfitter sets no
194
+ `CLAUDE_CONFIG_DIR`, and the composition reaches the session as a plugin directory instead. Claude
195
+ reads and writes `~/.claude` exactly as it does in a native session, so credentials, workspace
196
+ trust, permission approvals, and session history need no bridge at all — there is nothing to seed
197
+ and nothing to copy back, and nothing Outfitter does can race your other Claude sessions. The rest
198
+ of this section describes an **isolated** run (`--isolated`, or `isolation: isolated` in your
199
+ `~/.agents/settings.yml`), where the projection is the whole configuration.
200
+
201
+ Under isolation, Claude credentials need a narrow adapter bridge in addition to the path-keyed state below. Claude
194
202
  reads `.credentials.json` and `.claude.json` directly from `CLAUDE_CONFIG_DIR`; the ephemeral
195
203
  projection gives `.credentials.json` no durable home, and `.claude.json`'s native location
196
204
  (`~/.claude.json`, outside `~/.claude`) does not share its config-dir-relative path. Outfitter
@@ -203,11 +211,11 @@ instead of copying the projected credentials back. Claude MCP OAuth tokens
203
211
  live under `mcpOAuth` in `.credentials.json`, keyed by `<serverName>|<hash>`, so server
204
212
  authorizations acquired in an Outfitter-launched Claude session persist across runs through that
205
213
  whole-file copy-back. Outfitter never copies the full machine-local `~/.claude.json` into a
206
- projection or merges its other projected state back. Trust accepted inside an Outfitter session is
207
- therefore discarded, so Claude prompts for trust on every run in a workspace that was never trusted
208
- natively.
214
+ projection or merges its other projected state back. Trust accepted inside an isolated session is
215
+ therefore discarded, so an isolated run prompts for trust in a workspace that was never trusted
216
+ natively — which is one reason isolation is not the default.
209
217
 
210
- Claude session history has a second narrow bridge because `CLAUDE_CONFIG_DIR` also redirects
218
+ Isolated Claude session history has a second narrow bridge because `CLAUDE_CONFIG_DIR` also redirects
211
219
  Claude's native `projects/` tree into the temporary projection. Before launch, Outfitter derives
212
220
  Claude's project slug from the absolute working directory and copies only that slug directory from
213
221
  `~/.claude/projects/`. This keeps other projects' transcripts out of the projection while making
@@ -217,6 +225,9 @@ the projection's slug directories back into `~/.claude/projects/` atomically wit
217
225
  Durable files are never deleted. A seed or copy-back failure emits a warning and does not replace
218
226
  Claude's exit code or error.
219
227
 
228
+ These declared paths describe the isolated strategy; an inherited run writes to the native
229
+ locations directly and declares nothing.
230
+
220
231
  The Claude Code adapter declares these paths:
221
232
 
222
233
  ```yaml
@@ -46,13 +46,14 @@ Tasks and bake are not in this matrix — they are the subject of a [separate up
46
46
 
47
47
  ## Claude Code notes
48
48
 
49
- - **Config and session state** — Outfitter points `CLAUDE_CONFIG_DIR` at the baked composition. Before launch it copies only the current working directory's history from `~/.claude/projects/<project-slug>/` into the projection, so native `--continue` and `--resume` work without exposing other projects. After every successful or failed launch it atomically merges new or changed session files from every projected slug back into `~/.claude/projects/` with mode `0600`, never deleting durable history. Session bridge failures warn without masking the Claude exit. Outfitter also declares Claude state paths (`settings.json`, `agents/`, `skills/`, `commands/`, `plugins/`, `projects/`) for [state persistence](./state.md), and can [symlink a ported `~/.claude`](./porting-claude.md) so native use keeps working. MCP configuration from that port is no longer auto-discovered by Outfitter-launched Claude runs; those servers apply only when an agent selects them by slug. See the next bullet.
49
+ - **Your configuration comes first** — by default a Claude run stands on the configuration already on the machine. Outfitter sets no `CLAUDE_CONFIG_DIR`; it declares the baked composition a Claude plugin and passes it through `--plugin-dir`, so the session keeps your workspace trust, `~/.claude/settings.json` permissions, credentials, plugins, and configured MCP servers, and the profile's skills, subagents, and prompts layer on top. Nothing is seeded and nothing is copied back, because Claude is reading and writing its real configuration directory throughout. Pass `--isolated`, or set `isolation: isolated` in your `~/.agents/settings.yml`, to launch from the composition alone — the reproducible form for CI and containers, and what the remaining bullets in this section describe. If the installed Claude is too old to load a plugin directory, Outfitter falls back to an isolated run and says so rather than failing the launch.
50
+ - **Isolated config and session state** — an isolated run points `CLAUDE_CONFIG_DIR` at the baked composition. Before launch it copies only the current working directory's history from `~/.claude/projects/<project-slug>/` into the projection, so native `--continue` and `--resume` work without exposing other projects. After every successful or failed launch it atomically merges new or changed session files from every projected slug back into `~/.claude/projects/` with mode `0600`, never deleting durable history. Session bridge failures warn without masking the Claude exit. Outfitter also declares Claude state paths (`settings.json`, `agents/`, `skills/`, `commands/`, `plugins/`, `projects/`) for [state persistence](./state.md), and can [symlink a ported `~/.claude`](./porting-claude.md) so native use keeps working. MCP configuration from that port is no longer auto-discovered by Outfitter-launched Claude runs; those servers apply only when an agent selects them by slug. See the next bullet.
50
51
  - **Credentials, onboarding, and workspace trust** — before launch, Outfitter copies `~/.claude/.credentials.json` to the temporary root as `.credentials.json` with mode `0600`. The projected `.claude.json` contains `oauthAccount` and `hasCompletedOnboarding` when those keys are present in durable `~/.claude.json`. It also contains `projects[<cwd>].hasTrustDialogAccepted: true` only when that exact accepted trust decision already exists there; other projects and unrelated machine state are not copied. After any successful or failed launch, a `.credentials.json` changed by the run is copied back wholesale and `oauthAccount` is atomically merged into durable `.claude.json` without replacing unrelated keys. If the durable credentials also changed after seeding, Outfitter preserves that concurrent refresh and warns instead of copying back. MCP OAuth tokens live under `mcpOAuth` in `.credentials.json`, keyed by `<serverName>|<hash>`, so authorizations acquired in an Outfitter-launched Claude session persist across runs. Other projected `.claude.json` state, including trust accepted during the session, is discarded; a workspace that has never been trusted by native Claude therefore prompts again on every run.
51
- - **MCP servers** — every Claude launch passes the generated `mcp.json` through `--mcp-config` with `--strict-mcp-config`. MCP servers from user or project configuration, `.claude.json`, and plugins are therefore excluded; only servers selected by the composition are active.
52
- - **Subagents** — selected `agents/<id>` definitions are materialized into Claude's native agents directory.
52
+ - **MCP servers** — every Claude launch passes the generated `mcp.json` through `--mcp-config`. An inherited run stops there, so the composition's servers merge with the ones already configured on the machine: selecting a server says what the profile needs, not what the user may not have. An isolated run adds `--strict-mcp-config`, which excludes MCP servers from user or project configuration, `.claude.json`, and plugins so only the composition's servers are active.
53
+ - **Subagents** — selected `agents/<id>` definitions are materialized into the composition's agents directory. An inherited run loads them under the plugin's name (`<profile>:<subagent>`); an isolated run finds them natively under `CLAUDE_CONFIG_DIR`.
53
54
  - **Skills (Partial)** — selected skills are materialized into the config directory's skills surface; remaining gaps are tracked per release. The bundled Outfitter skill ships through the plugin channel.
54
55
  - **Model selection (Partial)** — model maps to `--model` and thinking level to `--effort`; provider selection is not projected for Claude and warns if requested.
55
- - **Hooks (Partial)** — hook configuration is projected into the generated `settings.json`; there is no portable protocol hooks resource yet. See [Hooks](./hooks.md).
56
+ - **Hooks** — Outfitter does not project hook configuration for Claude, and there is no portable protocol hooks resource yet. An inherited run keeps the hooks in your own `~/.claude/settings.json`; an isolated run has none. See [Hooks](./hooks.md).
56
57
  - **Tool availability** — `tools.allow` (after `tools.deny` removes entries) maps to both `--tools` (_availability_: an unlisted builtin is not in the session) and `--allowedTools` (_permission_: the granted tools are pre-approved, so a headless session is not stopped by a prompt); `tools.deny` always maps to `--disallowedTools`, including when both are declared, and a bare denied name removes the tool from context per Claude's docs. An allowlist that `tools.deny` empties maps to `--tools ""`, Claude's documented "disable all tools" form. Caveat: per the CLI reference, `--tools` governs the built-in set only — MCP tools (`mcp__server__*`) are unaffected and are governed by which MCP servers the loadout selects, so `--tools ""` is not exactly pi's zero-tool session when MCP servers are present. Claude's behavior here comes from `claude --help` and the CLI reference, not local measurement.
57
58
  - **DeepWork jobs** — job selection is Pi-only today and warns on Claude.
58
59
  - **Bundled Outfitter skill** — every launch also publishes Outfitter's own self-documentation skill as a bundled plugin, so the agent can explain Outfitter and this launch's configuration.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@ai-outfitter/outfitter",
3
- "version": "1.8.1",
3
+ "version": "1.10.0",
4
4
  "description": "Profile-oriented wrapper for launching pi, Claude Code, and future agent CLIs with reproducible configuration.",
5
5
  "type": "module",
6
6
  "repository": {
@@ -5,16 +5,16 @@
5
5
  "type": "object",
6
6
  "required": ["name"],
7
7
  "properties": {
8
- "name": { "type": "string", "pattern": "^[a-z0-9]+(?:-[a-z0-9]+)*$", "maxLength": 64 },
8
+ "name": { "$ref": "#/$defs/agentSlug" },
9
9
  "label": { "type": "string", "minLength": 1 },
10
10
  "description": { "type": "string" },
11
11
  "inherits": {
12
12
  "oneOf": [
13
- { "type": "string", "pattern": "^[a-z0-9]+(?:-[a-z0-9]+)*$", "maxLength": 64 },
13
+ { "$ref": "#/$defs/agentSlug" },
14
14
  {
15
15
  "type": "array",
16
16
  "minItems": 1,
17
- "items": { "type": "string", "pattern": "^[a-z0-9]+(?:-[a-z0-9]+)*$", "maxLength": 64 }
17
+ "items": { "$ref": "#/$defs/agentSlug" }
18
18
  }
19
19
  ],
20
20
  "description": "Parent agent slug or ordered parent slug list composed parent-first before this agent."
@@ -36,7 +36,7 @@
36
36
  "description": "Skill slugs resolved from agents/<name>/skills first, then catalog-wide skills."
37
37
  },
38
38
  "subagents": {
39
- "$ref": "#/$defs/slugList",
39
+ "$ref": "#/$defs/agentSlugList",
40
40
  "description": "Agent slugs delegated to. Resolved catalog-wide (a delegate is a shared agent, not nested under agents/<name>)."
41
41
  },
42
42
  "mcp": {
@@ -64,6 +64,15 @@
64
64
  },
65
65
  "additionalProperties": true,
66
66
  "$defs": {
67
+ "agentSlug": {
68
+ "type": "string",
69
+ "pattern": "^[a-z0-9]+(?:-[a-z0-9]+)*(?:\\.[a-z0-9]+(?:-[a-z0-9]+)*)*$",
70
+ "maxLength": 64
71
+ },
72
+ "agentSlugList": {
73
+ "type": "array",
74
+ "items": { "$ref": "#/$defs/agentSlug" }
75
+ },
67
76
  "toolName": {
68
77
  "type": "string",
69
78
  "minLength": 1,
@@ -6,6 +6,10 @@
6
6
  "properties": {
7
7
  "default_agent": { "type": "string", "minLength": 1 },
8
8
  "default_harness": { "enum": ["pi", "claude", "codex"] },
9
+ "isolation": {
10
+ "enum": ["inherit", "isolated"],
11
+ "description": "Whether a run stands on the machine's native harness configuration (inherit, the default) or on the projection alone (isolated). Honored only from home-scope settings."
12
+ },
9
13
  "cache_directory": { "type": "string", "minLength": 1 },
10
14
  "state_persistence": {
11
15
  "type": "object",