ruby_llm-contract 1.1.0 → 1.1.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +86 -0
- data/README.md +12 -3
- data/docs/guide/getting_started.md +17 -1
- data/docs/guide/relation_to_agent.md +8 -2
- data/lib/ruby_llm/contract/adapters/response.rb +28 -2
- data/lib/ruby_llm/contract/adapters/ruby_llm.rb +72 -18
- data/lib/ruby_llm/contract/concerns/eval_host.rb +2 -2
- data/lib/ruby_llm/contract/concerns/usage_aggregator.rb +8 -5
- data/lib/ruby_llm/contract/cost_calculator.rb +69 -41
- data/lib/ruby_llm/contract/eval/case_executor.rb +1 -1
- data/lib/ruby_llm/contract/eval/case_result.rb +3 -0
- data/lib/ruby_llm/contract/eval/eval_history.rb +2 -1
- data/lib/ruby_llm/contract/eval/evaluator/proc_evaluator.rb +6 -2
- data/lib/ruby_llm/contract/eval/model_comparison.rb +11 -1
- data/lib/ruby_llm/contract/eval/prompt_diff_serializer.rb +1 -1
- data/lib/ruby_llm/contract/eval/recommender.rb +12 -5
- data/lib/ruby_llm/contract/eval/report.rb +4 -1
- data/lib/ruby_llm/contract/eval/report_stats.rb +10 -5
- data/lib/ruby_llm/contract/eval/report_storage.rb +14 -8
- data/lib/ruby_llm/contract/eval/unknown_cost_gate.rb +0 -8
- data/lib/ruby_llm/contract/minitest.rb +2 -2
- data/lib/ruby_llm/contract/pipeline/runner.rb +8 -0
- data/lib/ruby_llm/contract/pipeline/trace.rb +10 -0
- data/lib/ruby_llm/contract/railtie.rb +1 -1
- data/lib/ruby_llm/contract/rake_task/suite_gate.rb +4 -4
- data/lib/ruby_llm/contract/rake_task.rb +4 -4
- data/lib/ruby_llm/contract/rspec/pass_eval.rb +2 -2
- data/lib/ruby_llm/contract/step/base.rb +21 -12
- data/lib/ruby_llm/contract/step/dsl.rb +5 -14
- data/lib/ruby_llm/contract/step/limit_checker.rb +2 -1
- data/lib/ruby_llm/contract/step/result_builder.rb +32 -3
- data/lib/ruby_llm/contract/step/retry_executor.rb +35 -4
- data/lib/ruby_llm/contract/step/runner.rb +7 -1
- data/lib/ruby_llm/contract/step/runner_config.rb +2 -2
- data/lib/ruby_llm/contract/step/trace.rb +54 -20
- data/lib/ruby_llm/contract/token_estimator.rb +5 -3
- data/lib/ruby_llm/contract/unknown_policy.rb +21 -0
- data/lib/ruby_llm/contract/version.rb +1 -1
- data/lib/ruby_llm/contract.rb +17 -4
- metadata +2 -1
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: 7ae19c7459f4999409748c604e8b80aa756e5b2333863c8cc16b4070bf57dff9
|
|
4
|
+
data.tar.gz: 3be220b3be5c0c313f4b3f5160da5416dc9be8c35629c0012d756e0d027aa646
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: 16bc0227706b8e1cc16f5d93c7771f82e7835c40c20888cec9163a9537d30f80a86f761c47fd10bd778df8f7023ce041bd0d574248950e11bdad5dd3ac4860da
|
|
7
|
+
data.tar.gz: 4d4374cc2d65aa555f30656c2475bae261b0af1be0656b7449c22c38467b7f1e32f7ba159d215adc17254d0c624f7144f91087644f578b5cb2727fc8f9087c91
|
data/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,91 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 1.1.2 (2026-10-10)
|
|
4
|
+
|
|
5
|
+
### Changed - costs and gates may move
|
|
6
|
+
|
|
7
|
+
- **Costs now count the prompt cache, and may go up.** RubyLLM 2.x reports cached prompt
|
|
8
|
+
tokens apart from `tokens.input`, and the adapter read only input and output: 1,000
|
|
9
|
+
uncached + 9,000 cached + 200 output tokens on gpt-4.1-mini came out at $0.00072
|
|
10
|
+
instead of $0.00162. `trace[:usage][:input_tokens]` is now the whole prompt and
|
|
11
|
+
`trace[:cost]` is the cost RubyLLM puts on the response: the price of the provider the
|
|
12
|
+
call went to, cache and long-context prices, the amount the provider reported when it
|
|
13
|
+
reports one, and RubyLLM's own retries of the request. Eval totals, history and
|
|
14
|
+
`compare_models` figures can rise against older baselines; baselines compare scores,
|
|
15
|
+
not costs, so they do not regress on this.
|
|
16
|
+
- **A call without token counts is no longer free.** When the provider sends no usage
|
|
17
|
+
(for the call or for any of RubyLLM's attempts at it), the counts still read 0 but
|
|
18
|
+
`trace[:usage_complete]` is `false` and, unless the provider reported what it billed,
|
|
19
|
+
the cost is unknown, so the eval cost gates refuse (opt out with
|
|
20
|
+
`on_unknown_pricing: :warn`).
|
|
21
|
+
- **A missing price is unknown, not $0.** A model priced for input but not output has no
|
|
22
|
+
cost instead of a partial one, so `max_cost` refuses it pre-flight. A call that wrote
|
|
23
|
+
to the prompt cache on a model without a cache-write price has an unknown cost
|
|
24
|
+
afterwards (pre-flight estimates do not predict cache writes).
|
|
25
|
+
|
|
26
|
+
### Fixed
|
|
27
|
+
|
|
28
|
+
- `CostCalculator` prices RubyLLM's models through `Model#cost_for`, so pre-flight
|
|
29
|
+
estimates use cache and long-context prices, and the price of the provider the call
|
|
30
|
+
goes to (12 model ids carry different prices per provider). `register_model` keeps
|
|
31
|
+
overriding the price for its id; cache reads are charged at its input price, cache
|
|
32
|
+
writes have no price.
|
|
33
|
+
- `compare_models` no longer names a model with unpriced cases the cheapest
|
|
34
|
+
(`best_for`), gives it an infinite `cost_per_point`, and `recommend` neither ranks it
|
|
35
|
+
on cost nor promises savings against it.
|
|
36
|
+
- `single_shot_cost` reported a later attempt's cost when the first attempt was unpriced.
|
|
37
|
+
- The step log printed an unknown cost as `$0.000000`; it now prints `cost=unknown`.
|
|
38
|
+
- Docs no longer say RubyLLM has no evaluation framework: `relation_to_agent.md` now
|
|
39
|
+
places `RubyLLM::Evaluation` (2.1) next to `define_eval`. The README and the getting
|
|
40
|
+
started guide describe what a trace's usage and cost count, and the README's 1.x
|
|
41
|
+
promise now says new trace keys may be added.
|
|
42
|
+
|
|
43
|
+
### Added
|
|
44
|
+
|
|
45
|
+
- `trace[:usage]` breakdown keys when reported: `cache_read_tokens`, `cache_write_tokens`,
|
|
46
|
+
`thinking_tokens` (parts of `input_tokens` / `output_tokens`, not extra tokens).
|
|
47
|
+
- `trace[:finish_reason]`, also on each retry attempt. A failed result whose response
|
|
48
|
+
stopped at a token limit or a content filter gets a last validation error saying so,
|
|
49
|
+
with the `max_output` the call ran with. Statuses and retries are unchanged.
|
|
50
|
+
- An adapter may report its own cost: `Adapters::Response.new(content:, usage:, cost:,
|
|
51
|
+
cost_complete:, usage_complete:, finish_reason:)`, where `cost: nil` means unknown.
|
|
52
|
+
A response without `cost:` is priced from `usage` as before.
|
|
53
|
+
- `provider:` on `CostCalculator.calculate`, `CostCalculator.find_model`,
|
|
54
|
+
`estimate_cost` and `estimate_eval_cost`.
|
|
55
|
+
- Baselines and history mark unpriced cases and runs with `cost_unknown: true`.
|
|
56
|
+
- A pipeline with a `token_budget` warns when a step's provider left out token counts;
|
|
57
|
+
`Pipeline::Trace#usage_complete?`.
|
|
58
|
+
|
|
59
|
+
## 1.1.1 (2026-10-10)
|
|
60
|
+
|
|
61
|
+
### Fixed
|
|
62
|
+
|
|
63
|
+
- **A failed rake eval gate printed "Eval suite FAILED: Eval suite FAILED".** The
|
|
64
|
+
message now reads "Eval suite FAILED: one or more evals did not pass".
|
|
65
|
+
|
|
66
|
+
### Changed (internal, no API change)
|
|
67
|
+
|
|
68
|
+
- Values that two or more places must agree on now have one owner, so one side can no
|
|
69
|
+
longer drift alone: the `:refuse`/`:warn` modes and their default for every
|
|
70
|
+
`on_unknown_*` option (`UnknownPolicy`), the Rails contract and eval dirs that the
|
|
71
|
+
Railtie ignores and `load_evals!` loads, the reload flag, the `:skipped` step status
|
|
72
|
+
(`CaseResult::SKIPPED_STATUS`), baseline and history file extensions, the context keys
|
|
73
|
+
forwarded to the adapter, the generic "passed"/"not passed" details the report hides,
|
|
74
|
+
and the optimize task's `MIN_SCORE` default, which now reads `Eval::DEFAULT_MIN_SCORE`.
|
|
75
|
+
The unknown-context-key warning lists the known keys in a different order, and in a
|
|
76
|
+
Rails app `load_evals!` now loads `app/contracts/eval` before `app/steps/eval` (the order
|
|
77
|
+
the Railtie and eager loading already used); this matters only if both dirs define an
|
|
78
|
+
eval of the same name on the same class, where the later file wins.
|
|
79
|
+
|
|
80
|
+
### Tests
|
|
81
|
+
|
|
82
|
+
- A real attachment through the RubyLLM 2.x adapter with HTTP intercepted (webmock):
|
|
83
|
+
pins the `input_image` part on the wire and the usage read back.
|
|
84
|
+
- A minimal Rails app booted in a child process: pins that Zeitwerk ignores the
|
|
85
|
+
`eval/` dirs, that evals register after boot, and that the rake tasks appear.
|
|
86
|
+
- `rubocop` is pinned to `~> 1.92.0`: CI resolves without a lockfile, so a floating
|
|
87
|
+
version let new cops land on CI before they ran locally.
|
|
88
|
+
|
|
3
89
|
## 1.1.0 (2026-10-10)
|
|
4
90
|
|
|
5
91
|
### Changed - may turn a green CI red
|
data/README.md
CHANGED
|
@@ -91,6 +91,8 @@ result.trace[:attempts]
|
|
|
91
91
|
# ]
|
|
92
92
|
```
|
|
93
93
|
|
|
94
|
+
`usage[:input_tokens]` is the whole prompt, prompt-cache tokens included (broken out as `cache_read_tokens` / `cache_write_tokens` when the provider reports them), and `cost` is what RubyLLM prices the response at: the provider's price, cache and long-context rates, or the amount the provider reported. A cost the gem cannot fully price is marked unknown rather than counted as $0 - see [what a trace's usage and cost count](docs/guide/getting_started.md#what-a-traces-usage-and-cost-count).
|
|
95
|
+
|
|
94
96
|
If the response is malformed, the TL;DR overflows the card, or the takeaway count is off, the gem moves to the next step. This is model **escalation**, not a fallback list — each step is an independent config (`model`, `reasoning_effort`), so the retry policy spends more compute only when the cheaper one couldn't satisfy the contract.
|
|
95
97
|
|
|
96
98
|
### Add a CI gate in 6 lines
|
|
@@ -128,6 +130,7 @@ Everything below is optional — the example above is a complete step. Reach for
|
|
|
128
130
|
- **[Find the cheapest viable fallback list](docs/guide/optimizing_retry_policy.md)** — empirically pick the cheapest model chain that still passes your evals.
|
|
129
131
|
- **[A/B test prompts](docs/guide/eval_first.md)** — measure whether a new prompt is safe to ship before merging.
|
|
130
132
|
- **[Budget caps](docs/guide/getting_started.md)** — refuse the request pre-flight when an estimate exceeds the limit.
|
|
133
|
+
- **[Cost tracking](docs/guide/getting_started.md#what-a-traces-usage-and-cost-count)** - per-call cost with prompt-cache and per-provider prices; eval cost gates fail closed when a cost is unknown.
|
|
131
134
|
- **[Reasoning effort / thinking config](docs/guide/optimizing_retry_policy.md)** — Anthropic / OpenAI thinking configuration on the Step class.
|
|
132
135
|
|
|
133
136
|
Also supports [multi-step pipelines](docs/guide/pipeline.md) with fail-fast and per-step models.
|
|
@@ -136,7 +139,7 @@ Also supports [multi-step pipelines](docs/guide/pipeline.md) with fail-fast and
|
|
|
136
139
|
|
|
137
140
|
## Relation to `RubyLLM::Agent`
|
|
138
141
|
|
|
139
|
-
`Step::Base` and `RubyLLM::Agent` (since RubyLLM 1.12) are **siblings** targeting the same niche: reusable, class-based prompts. Both call into `RubyLLM::Chat` directly
|
|
142
|
+
`Step::Base` and `RubyLLM::Agent` (since RubyLLM 1.12) are **siblings** targeting the same niche: reusable, class-based prompts. Both call into `RubyLLM::Chat` directly - Step does not wrap Agent. Step adds the contract layer: `validate` (business invariants), `retry_policy escalate(...)` (model escalation on validation failure), `max_cost` pre-flight refusal, regression-eval framework, pipeline composition. RubyLLM 2.1's `RubyLLM::Evaluation` grades answers; `define_eval` guards against regressions and cost in CI - the mapping covers both. **[Full feature mapping →](docs/guide/relation_to_agent.md)**
|
|
140
143
|
|
|
141
144
|
## Relation to `ruby_llm-tribunal`
|
|
142
145
|
|
|
@@ -153,7 +156,7 @@ Different layers, complementary. [`ruby_llm-tribunal`](https://github.com/Alqemi
|
|
|
153
156
|
| [Why contracts?](docs/guide/why.md) | Recognise the four production failures the gem exists for |
|
|
154
157
|
| [Relation to RubyLLM::Agent](docs/guide/relation_to_agent.md) | Sibling abstractions; what each adds; runtime call path; coexistence patterns |
|
|
155
158
|
| [Relation to ruby_llm-tribunal](docs/guide/relation_to_tribunal.md) | Different layers (test framework vs runtime contract); visual flows; integration recipes |
|
|
156
|
-
| [Getting Started](docs/guide/getting_started.md) | Walk the full feature set on one concrete step |
|
|
159
|
+
| [Getting Started](docs/guide/getting_started.md) | Walk the full feature set on one concrete step, including what usage and cost count |
|
|
157
160
|
| [Rails integration](docs/guide/rails_integration.md) | Directory, initializer, jobs, logging, specs, CI gate — 7 FAQs for Rails devs |
|
|
158
161
|
| [Adopt in an existing Rails app](docs/guide/migration.md) | Replace raw `LlmClient.call` with a contract, Before/After |
|
|
159
162
|
| [Prevent silent prompt regressions](docs/guide/eval_first.md) | Evals, baselines, CI gates that block quality drift |
|
|
@@ -173,7 +176,11 @@ Stable since **1.0.0**. Semver tracked; breaking changes flagged in [CHANGELOG](
|
|
|
173
176
|
What 1.0 promises not to break inside the 1.x line: the documented DSL
|
|
174
177
|
(`prompt`, `output_schema`, `validate`, `retry_policy`, `max_cost`, `define_eval`,
|
|
175
178
|
pipelines), the `Result` and `trace` shapes including `trace[:usage]`'s
|
|
176
|
-
`{ input_tokens:, output_tokens: }` keys, and the adapter interface.
|
|
179
|
+
`{ input_tokens:, output_tokens: }` keys, and the adapter interface. New keys may be
|
|
180
|
+
added to both (1.1.2 added `cache_read_tokens`, `cache_write_tokens` and
|
|
181
|
+
`thinking_tokens` to `trace[:usage]`, and `usage_complete`, `cost_complete` and
|
|
182
|
+
`finish_reason` to the trace), and a figure may change when it was wrong: since 1.1.2
|
|
183
|
+
`input_tokens` counts prompt-cache tokens. A new major
|
|
177
184
|
version of a runtime dependency may require a new major version here — that is
|
|
178
185
|
what 1.0.0 itself was.
|
|
179
186
|
|
|
@@ -185,6 +192,8 @@ what 1.0.0 itself was.
|
|
|
185
192
|
|
|
186
193
|
**Where in a Rails app?** Default `app/contracts/`. The Railtie reloads `app/contracts/eval/` and `app/steps/eval/` in development; any autoloaded directory also works. See [Rails integration](docs/guide/rails_integration.md).
|
|
187
194
|
|
|
195
|
+
**Costs went up, or an eval cost gate turned red, after upgrading to 1.1.2?** Earlier versions left prompt-cache tokens out of `input_tokens` and the cost, priced every model at its default provider, counted a missing price as $0, and treated a call without token counts as free. The [CHANGELOG](CHANGELOG.md) lists each change; `on_unknown_pricing: :warn` turns the unknown-cost refusal into a warning.
|
|
196
|
+
|
|
188
197
|
**Upgraded from pre-0.10.0 and getting `:limit_exceeded` with attachments?** Multimodal contracts with `max_cost`/`max_input` need `attachment_token_estimate`. See [multimodal input guide](docs/guide/multimodal_input.md#cost-attachment_token_estimate-is-required) for setup, fail-closed behaviour, and `on_unknown_attachment_size :warn` opt-out.
|
|
189
198
|
|
|
190
199
|
## License
|
|
@@ -173,7 +173,23 @@ max_cost 0.01, on_unknown_pricing: :warn
|
|
|
173
173
|
|
|
174
174
|
Default is `:refuse`. Use `:warn` only when you accept running without a cost ceiling (fine-tuned models you trust, private endpoints).
|
|
175
175
|
|
|
176
|
-
The eval-level gates - `with_maximum_cost` on `pass_eval`, `maximum_cost` on the rake task, and `maximum_cost:` on `assert_eval_passes` - follow the same rule since 1.1.0. A report's total counts a case it cannot price as $0, so when any case ran on a model without pricing data the gate fails and names the cases instead of comparing an undercounted total. Opt out with `.on_unknown_pricing(:warn)`, `t.on_unknown_pricing = :warn` or `on_unknown_pricing: :warn`, which checks the budget against the priced cases and prints a warning. Offline runs (`sample_response`, a `Test` adapter without `usage:`) report zero tokens, cost $0 with or without pricing, and are never flagged.
|
|
176
|
+
The eval-level gates - `with_maximum_cost` on `pass_eval`, `maximum_cost` on the rake task, and `maximum_cost:` on `assert_eval_passes` - follow the same rule since 1.1.0. A report's total counts a case it cannot price as $0, so when any case ran on a model without pricing data the gate fails and names the cases instead of comparing an undercounted total. Opt out with `.on_unknown_pricing(:warn)`, `t.on_unknown_pricing = :warn` or `on_unknown_pricing: :warn`, which checks the budget against the priced cases and prints a warning. Offline runs (`sample_response`, a `Test` adapter without `usage:`) report zero tokens, cost $0 with or without pricing, and are never flagged. Since 1.1.2 a real call whose provider sent no token counts is flagged too, instead of passing as a free zero-token run.
|
|
177
|
+
|
|
178
|
+
### What a trace's usage and cost count
|
|
179
|
+
|
|
180
|
+
`trace[:usage][:input_tokens]` is the whole prompt the model took in, including tokens served from or written to the provider's prompt cache. When the provider reports them, the trace adds a breakdown: `cache_read_tokens`, `cache_write_tokens` and `thinking_tokens`. These are parts of `input_tokens` / `output_tokens`, not extra tokens; reasoning tokens are already in `output_tokens` when the provider bills them as output, and a model with a separate reasoning price is charged that price for them.
|
|
181
|
+
|
|
182
|
+
```ruby
|
|
183
|
+
result.trace[:usage]
|
|
184
|
+
# => { input_tokens: 10_000, output_tokens: 200, cache_read_tokens: 9_000 }
|
|
185
|
+
result.trace[:cost] # => 0.00162 (gpt-4.1-mini: cache reads at the cache price)
|
|
186
|
+
```
|
|
187
|
+
|
|
188
|
+
With the RubyLLM adapter, `trace[:cost]` is the cost RubyLLM puts on the response (`response.cost.total`): the price of the provider the call went to, cache and long-context prices, and the amount the provider reported when it reports one (OpenRouter, for example). A price set with `CostCalculator.register_model` still overrides it for that model id, unless the provider reported what it billed; there, cache reads are charged at the input price and a call that wrote to the cache has no price, since cache writes can cost more than input.
|
|
189
|
+
|
|
190
|
+
When a cost does not cover the whole call, `result.trace.cost_unknown?` is true and the eval gates above refuse. That happens when pricing is missing for a token category the call used, or when the provider sent no token counts for the call or for one of RubyLLM's attempts at it (`trace[:usage_complete]` is then `false` and missing counts read as 0) and did not report what it billed either. On a retried step `trace[:cost]` is the sum of the attempts that could be priced, and `cost_unknown?` is true if any attempt could not. `max_cost` itself is a pre-flight check on an estimate; it does not inspect the finished call.
|
|
191
|
+
|
|
192
|
+
A custom adapter returning `Response.new(content:, usage:)` is priced from `usage` as before. To report its own cost, pass `cost:` - `nil` means unknown - with `cost_complete:` and `usage_complete:`.
|
|
177
193
|
|
|
178
194
|
### Preflight cost estimates
|
|
179
195
|
|
|
@@ -12,8 +12,8 @@
|
|
|
12
12
|
| `validate :rule do ... end` business invariants on output | only in `ruby_llm-contract` |
|
|
13
13
|
| `retry_policy escalate(...)` model escalation on validation failure | only here (different from RubyLLM's network-level retry) |
|
|
14
14
|
| `max_cost` / `max_input` / `max_output` pre-flight refusal | only here |
|
|
15
|
-
| `define_eval` + baseline regression + `compare_models` + `optimize_retry_policy` | only here
|
|
16
|
-
| Pipeline composition with `step SomeStep, as: :alias` | only here
|
|
15
|
+
| `define_eval` + baseline regression + `compare_models` + `optimize_retry_policy` | only here; RubyLLM 2.1's `RubyLLM::Evaluation` grades answers (see below) but keeps no baseline and has no cost gate |
|
|
16
|
+
| Pipeline composition with `step SomeStep, as: :alias` | only here; `RubyLLM.workflow` (2.1) groups calls for tracing, without typed step outputs, fail-fast or budgets |
|
|
17
17
|
| `around_call`, named `observe` hooks with pass/fail recorded in trace | only here |
|
|
18
18
|
|
|
19
19
|
## Runtime relationship
|
|
@@ -32,6 +32,12 @@ Step.run(input)
|
|
|
32
32
|
|
|
33
33
|
This may change in a future release if upstream APIs make a layered design natural. The decision is not committed; it depends on adopter signal.
|
|
34
34
|
|
|
35
|
+
## Relation to `RubyLLM::Evaluation`
|
|
36
|
+
|
|
37
|
+
RubyLLM 2.1 ships `RubyLLM::Evaluation`: a dataset of cases run through your `perform` method, graded for correctness by an LLM or by your own criteria and Ruby assertions, with RSpec/Minitest integration and a report of verdicts, tokens and cost. Use it to answer "is this answer right?".
|
|
38
|
+
|
|
39
|
+
`define_eval` answers a different question: did this prompt or model get worse than last time, and does it stay within budget? It keeps baselines and flags regressions, compares prompts (`compare_with`) and models (`compare_models`, `recommend`), runs offline with `sample_response`, and gates CI on score and cost. The two can share a project; an Evaluation's `perform` can call `Step.run`.
|
|
40
|
+
|
|
35
41
|
## Coexistence on the same project
|
|
36
42
|
|
|
37
43
|
The two abstractions can live in the same Rails (or non-Rails) project. Pick one per use case:
|
|
@@ -3,16 +3,42 @@
|
|
|
3
3
|
module RubyLLM
|
|
4
4
|
module Contract
|
|
5
5
|
module Adapters
|
|
6
|
+
# `content` and `usage` are the whole interface an adapter has to fill.
|
|
7
|
+
# The rest is optional accounting: an adapter that leaves `cost` out gets
|
|
8
|
+
# its usage priced from the registry, as before; one that passes `cost`
|
|
9
|
+
# (nil included) owns the figure, and `cost_complete` / `usage_complete`
|
|
10
|
+
# say whether it and the token counts cover the whole call.
|
|
6
11
|
class Response
|
|
7
12
|
include Concerns::DeepFreeze
|
|
8
13
|
|
|
9
|
-
attr_reader :content, :usage
|
|
14
|
+
attr_reader :content, :usage, :usage_complete, :cost_complete, :finish_reason
|
|
10
15
|
|
|
11
|
-
def initialize(content:, usage: {}
|
|
16
|
+
def initialize(content:, usage: {}, cost: Step::Trace::COST_UNSET, usage_complete: nil,
|
|
17
|
+
cost_complete: nil, finish_reason: nil)
|
|
12
18
|
@content = deep_dup_freeze(content)
|
|
13
19
|
@usage = deep_dup_freeze(usage)
|
|
20
|
+
@cost = cost
|
|
21
|
+
@usage_complete = usage_complete
|
|
22
|
+
@cost_complete = cost_complete
|
|
23
|
+
@finish_reason = finish_reason
|
|
14
24
|
freeze
|
|
15
25
|
end
|
|
26
|
+
|
|
27
|
+
def cost
|
|
28
|
+
cost_provided? ? @cost : nil
|
|
29
|
+
end
|
|
30
|
+
|
|
31
|
+
def cost_provided?
|
|
32
|
+
!@cost.equal?(Step::Trace::COST_UNSET)
|
|
33
|
+
end
|
|
34
|
+
|
|
35
|
+
# Keyword arguments for Step::Trace.new. Omits what the adapter did not
|
|
36
|
+
# report, so a plain Response keeps the registry pricing path.
|
|
37
|
+
def trace_accounting
|
|
38
|
+
fields = { usage_complete: @usage_complete, cost_complete: @cost_complete,
|
|
39
|
+
finish_reason: @finish_reason }.compact
|
|
40
|
+
cost_provided? ? fields.merge(cost: @cost) : fields
|
|
41
|
+
end
|
|
16
42
|
end
|
|
17
43
|
end
|
|
18
44
|
end
|
|
@@ -13,16 +13,13 @@ module RubyLLM
|
|
|
13
13
|
chat = build_chat(options, system_contents)
|
|
14
14
|
add_history(chat, conversation[0..-2])
|
|
15
15
|
|
|
16
|
-
# `with: nil`
|
|
17
|
-
#
|
|
18
|
-
# `Content.new(text, nil)` keeps text-only path when attachments
|
|
19
|
-
# are empty; raise only fires when BOTH text and attachments are nil,
|
|
20
|
-
# and we always pass a non-nil string thanks to `&.fetch(:content, "")`).
|
|
16
|
+
# `with: nil` adds no attachment. The text is never nil (`fetch(:content, "")`),
|
|
17
|
+
# and RubyLLM raises only when both text and attachments are nil.
|
|
21
18
|
response = chat.ask(
|
|
22
19
|
conversation.last&.fetch(:content, ""),
|
|
23
20
|
with: options[:attachment]
|
|
24
21
|
)
|
|
25
|
-
build_response(response)
|
|
22
|
+
build_response(response, options[:model])
|
|
26
23
|
end
|
|
27
24
|
|
|
28
25
|
CHAT_OPTION_METHODS = {
|
|
@@ -64,8 +61,7 @@ module RubyLLM
|
|
|
64
61
|
# taking precedence over `:thinking[:effort]`. This is the per-attempt
|
|
65
62
|
# override path used by `retry_policy { escalate({model:, reasoning_effort:}) }`
|
|
66
63
|
# — the attempt-specific effort must win over the class-level default.
|
|
67
|
-
# Forwarded provider-agnostically via `chat.with_thinking(**)
|
|
68
|
-
# available since RubyLLM 1.12 (gemspec enforces this minimum).
|
|
64
|
+
# Forwarded provider-agnostically via `chat.with_thinking(**)`.
|
|
69
65
|
thinking_config = resolve_thinking_config(options)
|
|
70
66
|
chat.with_thinking(**thinking_config) if thinking_config
|
|
71
67
|
|
|
@@ -84,26 +80,84 @@ module RubyLLM
|
|
|
84
80
|
base.empty? ? nil : base
|
|
85
81
|
end
|
|
86
82
|
|
|
87
|
-
def build_response(response)
|
|
83
|
+
def build_response(response, model)
|
|
88
84
|
content = response.content
|
|
89
85
|
content = content.to_s unless content.is_a?(Hash) || content.is_a?(Array)
|
|
90
86
|
|
|
91
|
-
# This is the ONLY place upstream token counts are read.
|
|
92
|
-
#
|
|
93
|
-
#
|
|
94
|
-
#
|
|
95
|
-
# documented in the README) and deliberately does NOT change.
|
|
87
|
+
# This is the ONLY place upstream token counts are read. The
|
|
88
|
+
# `{ input_tokens:, output_tokens: }` shape is this gem's public
|
|
89
|
+
# contract (Step::Trace#usage, README) and does not change; RubyLLM's
|
|
90
|
+
# own counts go into it as described in `usage_from`.
|
|
96
91
|
tokens = response.tokens
|
|
92
|
+
usage_complete = every_attempt_counted?(response, tokens)
|
|
93
|
+
reported = tokens&.reported_cost
|
|
94
|
+
usage = usage_from(tokens)
|
|
95
|
+
cost = call_cost(response, model, usage, reported)
|
|
97
96
|
|
|
98
97
|
Response.new(
|
|
99
98
|
content: content,
|
|
100
|
-
usage:
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
99
|
+
usage: usage,
|
|
100
|
+
cost: cost,
|
|
101
|
+
usage_complete: usage_complete,
|
|
102
|
+
# A cost RubyLLM priced from partial counts covers part of the call;
|
|
103
|
+
# one the provider reported covers all of it.
|
|
104
|
+
cost_complete: !cost.nil? && (usage_complete || !reported.nil?),
|
|
105
|
+
finish_reason: response.finish_reason
|
|
104
106
|
)
|
|
105
107
|
end
|
|
106
108
|
|
|
109
|
+
# `response.tokens` sums RubyLLM's attempts at the request, skipping
|
|
110
|
+
# counts an attempt never got, so the sum alone can look complete.
|
|
111
|
+
# RubyLLM records each attempt; one that failed after the provider may
|
|
112
|
+
# have billed it keeps nil counts, one never sent or refused gets zeros.
|
|
113
|
+
def every_attempt_counted?(response, tokens)
|
|
114
|
+
entries = response.respond_to?(:ruby_llm_usage_entries) ? Array(response.ruby_llm_usage_entries) : []
|
|
115
|
+
return entries.all? { |entry| reported_both_counts?(entry.tokens) } if entries.any?
|
|
116
|
+
|
|
117
|
+
reported_both_counts?(tokens)
|
|
118
|
+
end
|
|
119
|
+
|
|
120
|
+
# Zero is a reported count (a full cache hit leaves uncached input at 0);
|
|
121
|
+
# nil is a count the provider did not send.
|
|
122
|
+
def reported_both_counts?(tokens)
|
|
123
|
+
!tokens.nil? && !tokens.input.nil? && !tokens.output.nil?
|
|
124
|
+
end
|
|
125
|
+
|
|
126
|
+
# RubyLLM 2.x counts `input` without the prompt-cache tokens, which it
|
|
127
|
+
# reports apart as `cache_read` / `cache_write`. Our input_tokens is the
|
|
128
|
+
# whole prompt, so they are added back and also kept as a breakdown.
|
|
129
|
+
# `output` already includes thinking tokens when the provider bills them
|
|
130
|
+
# as output, so `thinking_tokens` is a breakdown, never added. A count
|
|
131
|
+
# the provider did not report is 0 here; `usage_complete` says so.
|
|
132
|
+
def usage_from(tokens)
|
|
133
|
+
return { input_tokens: 0, output_tokens: 0 } unless tokens
|
|
134
|
+
|
|
135
|
+
cache_read = tokens.cache_read.to_i
|
|
136
|
+
cache_write = tokens.cache_write.to_i
|
|
137
|
+
usage = { input_tokens: tokens.input.to_i + cache_read + cache_write, output_tokens: tokens.output.to_i }
|
|
138
|
+
usage[:cache_read_tokens] = cache_read if cache_read.positive?
|
|
139
|
+
usage[:cache_write_tokens] = cache_write if cache_write.positive?
|
|
140
|
+
usage[:thinking_tokens] = tokens.thinking.to_i if tokens.thinking.to_i.positive?
|
|
141
|
+
usage
|
|
142
|
+
end
|
|
143
|
+
|
|
144
|
+
# RubyLLM prices the response itself: per provider, with cache and
|
|
145
|
+
# long-context prices, the amount the provider reported when it did,
|
|
146
|
+
# and its own retries of the request. It returns nil when it cannot
|
|
147
|
+
# price every attempt. A price set with `register_model` still wins
|
|
148
|
+
# for its id, unless the provider reported what it billed; it prices
|
|
149
|
+
# the summed counts, so it is complete only when every attempt was
|
|
150
|
+
# counted (`cost_complete` in build_response).
|
|
151
|
+
def call_cost(response, model, usage, reported)
|
|
152
|
+
if reported.nil? && CostCalculator.registered?(model)
|
|
153
|
+
CostCalculator.calculate(model_name: model, usage: usage)
|
|
154
|
+
else
|
|
155
|
+
response.cost.total&.round(6)
|
|
156
|
+
end
|
|
157
|
+
rescue StandardError
|
|
158
|
+
nil
|
|
159
|
+
end
|
|
160
|
+
|
|
107
161
|
def partition_messages(messages)
|
|
108
162
|
system_contents = []
|
|
109
163
|
conversation = []
|
|
@@ -16,13 +16,13 @@ module RubyLLM
|
|
|
16
16
|
@file_sourced_evals ||= Set.new
|
|
17
17
|
key = name.to_s
|
|
18
18
|
|
|
19
|
-
if @eval_definitions.key?(key) && !
|
|
19
|
+
if @eval_definitions.key?(key) && !Contract.reloading?
|
|
20
20
|
warn "[ruby_llm-contract] Redefining eval '#{key}' on #{self}. " \
|
|
21
21
|
"This replaces the previous definition."
|
|
22
22
|
end
|
|
23
23
|
|
|
24
24
|
@eval_definitions[key] = Eval::EvalDefinition.new(key, step_class: self, &)
|
|
25
|
-
@file_sourced_evals.add(key) if
|
|
25
|
+
@file_sourced_evals.add(key) if Contract.reloading?
|
|
26
26
|
Contract.register_eval_host(self)
|
|
27
27
|
register_subclasses(self)
|
|
28
28
|
end
|
|
@@ -4,6 +4,10 @@ module RubyLLM
|
|
|
4
4
|
module Contract
|
|
5
5
|
module Concerns
|
|
6
6
|
module UsageAggregator
|
|
7
|
+
# Breakdown keys an adapter may add next to input/output. They are parts
|
|
8
|
+
# of input_tokens/output_tokens, not extra tokens, so sum_tokens skips them.
|
|
9
|
+
USAGE_DETAIL_KEYS = %i[cache_read_tokens cache_write_tokens thinking_tokens].freeze
|
|
10
|
+
|
|
7
11
|
private
|
|
8
12
|
|
|
9
13
|
def extract_usage(trace_entry)
|
|
@@ -24,18 +28,17 @@ module RubyLLM
|
|
|
24
28
|
end
|
|
25
29
|
|
|
26
30
|
def aggregate_usage(traces)
|
|
27
|
-
|
|
28
|
-
output_total = 0
|
|
31
|
+
totals = Hash.new(0)
|
|
29
32
|
|
|
30
33
|
traces.each do |trace_entry|
|
|
31
34
|
usage = extract_usage(trace_entry)
|
|
32
35
|
next unless usage.is_a?(Hash)
|
|
33
36
|
|
|
34
|
-
|
|
35
|
-
output_total += usage[:output_tokens] || 0
|
|
37
|
+
[:input_tokens, :output_tokens, *USAGE_DETAIL_KEYS].each { |key| totals[key] += usage[key] || 0 }
|
|
36
38
|
end
|
|
37
39
|
|
|
38
|
-
{ input_tokens:
|
|
40
|
+
{ input_tokens: totals[:input_tokens], output_tokens: totals[:output_tokens] }
|
|
41
|
+
.merge(totals.slice(*USAGE_DETAIL_KEYS).select { |_, count| count.positive? })
|
|
39
42
|
end
|
|
40
43
|
end
|
|
41
44
|
end
|
|
@@ -7,20 +7,22 @@ module RubyLLM
|
|
|
7
7
|
# **What this module does (public surface):**
|
|
8
8
|
#
|
|
9
9
|
# 1. **Fine-tune / custom-model pricing registry** — `register_model`
|
|
10
|
-
# fills
|
|
11
|
-
# upstream
|
|
12
|
-
# (e.g. `ft:gpt-4o-custom`) need their pricing supplied
|
|
10
|
+
# fills a gap RubyLLM still has in 2.x: there is no simple
|
|
11
|
+
# upstream API to price a model its registry lacks, so fine-tuned
|
|
12
|
+
# models (e.g. `ft:gpt-4o-custom`) need their pricing supplied
|
|
13
|
+
# locally. A registered price also overrides RubyLLM's for that id.
|
|
13
14
|
# 2. **Lookup with fallback chain** — `calculate(model_name:, usage:)`
|
|
14
15
|
# checks the custom registry first, falls back to
|
|
15
16
|
# `RubyLLM.models.find(model_name)`, returns `nil` on miss.
|
|
16
17
|
#
|
|
17
18
|
# **What this module is NOT:**
|
|
18
19
|
#
|
|
19
|
-
# - Not a
|
|
20
|
-
#
|
|
21
|
-
#
|
|
22
|
-
#
|
|
23
|
-
#
|
|
20
|
+
# - Not a substitute for RubyLLM's pricing — for any model in
|
|
21
|
+
# `RubyLLM.models`, RubyLLM's own `Model#cost_for` does the math.
|
|
22
|
+
# - Not the source of a finished call's cost: the RubyLLM adapter takes
|
|
23
|
+
# that from the response (`response.cost`), which also covers what
|
|
24
|
+
# the provider reported and RubyLLM's own retries. This module prices
|
|
25
|
+
# pre-flight estimates and calls from adapters that report no cost.
|
|
24
26
|
#
|
|
25
27
|
# The reason this module exists at all is the registry + retry usage
|
|
26
28
|
# aggregation across attempts (the latter sits in `Step::RetryExecutor`,
|
|
@@ -60,8 +62,9 @@ module RubyLLM
|
|
|
60
62
|
|
|
61
63
|
# Look up cost for a single model + usage hash.
|
|
62
64
|
# Returns nil if model is unknown (custom registry miss + RubyLLM miss),
|
|
63
|
-
#
|
|
64
|
-
#
|
|
65
|
+
# or if any used token category has no price, so callers can decide
|
|
66
|
+
# whether to refuse the call or proceed (see `on_unknown_pricing:` step
|
|
67
|
+
# option for the budget-gating policy).
|
|
65
68
|
#
|
|
66
69
|
# CostCalculator.calculate(
|
|
67
70
|
# model_name: "gpt-4o-mini",
|
|
@@ -69,13 +72,17 @@ module RubyLLM
|
|
|
69
72
|
# )
|
|
70
73
|
# # => 0.00069 (or nil if model not registered)
|
|
71
74
|
#
|
|
72
|
-
#
|
|
73
|
-
#
|
|
74
|
-
#
|
|
75
|
-
|
|
75
|
+
# `usage[:input_tokens]` is the whole prompt; `:cache_read_tokens` and
|
|
76
|
+
# `:cache_write_tokens`, when present, are the parts of it served from or
|
|
77
|
+
# written to the provider's prompt cache. `provider:` picks the provider's
|
|
78
|
+
# price for a model id several providers list at different prices.
|
|
79
|
+
#
|
|
80
|
+
# Aggregating across retry attempts is done in `Step::RetryExecutor`,
|
|
81
|
+
# not here.
|
|
82
|
+
def self.calculate(model_name:, usage:, provider: nil)
|
|
76
83
|
return nil unless model_name && usage.is_a?(Hash)
|
|
77
84
|
|
|
78
|
-
model_info = find_model(model_name)
|
|
85
|
+
model_info = find_model(model_name, provider: provider)
|
|
79
86
|
return nil unless model_info
|
|
80
87
|
|
|
81
88
|
compute_cost(model_info, usage)
|
|
@@ -83,55 +90,76 @@ module RubyLLM
|
|
|
83
90
|
nil
|
|
84
91
|
end
|
|
85
92
|
|
|
86
|
-
def self.
|
|
87
|
-
|
|
88
|
-
return nil unless prices
|
|
89
|
-
|
|
90
|
-
input_cost = token_cost(usage[:input_tokens], prices[:input])
|
|
91
|
-
output_cost = token_cost(usage[:output_tokens], prices[:output])
|
|
92
|
-
(input_cost + output_cost).round(6)
|
|
93
|
+
def self.registered?(model_name)
|
|
94
|
+
@custom_models.key?(model_name)
|
|
93
95
|
end
|
|
94
96
|
|
|
95
97
|
# Two shapes reach here. `RegisteredModel` (our own struct, from
|
|
96
|
-
# `register_model`) exposes flat `*_price_per_million` readers
|
|
97
|
-
#
|
|
98
|
-
#
|
|
98
|
+
# `register_model`) exposes flat `*_price_per_million` readers and is
|
|
99
|
+
# priced here. A RubyLLM model is priced by RubyLLM itself
|
|
100
|
+
# (`Model#cost_for`), which knows cache, long-context and per-provider
|
|
101
|
+
# prices and returns nil when a used category has no price.
|
|
99
102
|
#
|
|
100
|
-
# nil means "unknown pricing".
|
|
101
|
-
#
|
|
102
|
-
#
|
|
103
|
-
#
|
|
104
|
-
#
|
|
105
|
-
def self.
|
|
103
|
+
# nil means "unknown pricing". `calculate` rescues StandardError, so a
|
|
104
|
+
# future shape neither branch can read also degrades to nil: `max_cost`
|
|
105
|
+
# then refuses the call citing "no pricing data" rather than the real
|
|
106
|
+
# cause, and paths that floor unknown cost to 0.0 (eval cost estimates,
|
|
107
|
+
# report totals) undercount.
|
|
108
|
+
def self.compute_cost(model_info, usage)
|
|
106
109
|
if model_info.respond_to?(:input_price_per_million)
|
|
107
|
-
|
|
108
|
-
|
|
110
|
+
registered_cost(model_info, usage)
|
|
111
|
+
elsif model_info.respond_to?(:cost_for)
|
|
112
|
+
model_info.cost_for(tokens_for(usage)).total&.round(6)
|
|
109
113
|
end
|
|
114
|
+
end
|
|
110
115
|
|
|
111
|
-
|
|
112
|
-
|
|
116
|
+
# Our registry knows input and output prices only. Cache reads are
|
|
117
|
+
# charged at the input price, which assumes they cost no more than
|
|
118
|
+
# uncached input (true of every provider in RubyLLM's registry, not
|
|
119
|
+
# guaranteed for a custom one). Cache writes can cost more than input
|
|
120
|
+
# (Anthropic charges 1.25x-2x), so a call that wrote to the cache has
|
|
121
|
+
# no price here rather than an undercount.
|
|
122
|
+
def self.registered_cost(model_info, usage)
|
|
123
|
+
return nil if (usage[:cache_write_tokens] || 0).positive?
|
|
124
|
+
|
|
125
|
+
input_cost = token_cost(usage[:input_tokens], model_info.input_price_per_million)
|
|
126
|
+
output_cost = token_cost(usage[:output_tokens], model_info.output_price_per_million)
|
|
127
|
+
(input_cost + output_cost).round(6)
|
|
128
|
+
end
|
|
113
129
|
|
|
114
|
-
|
|
130
|
+
# RubyLLM counts uncached input apart from cache reads and writes; our
|
|
131
|
+
# usage counts them all in input_tokens.
|
|
132
|
+
def self.tokens_for(usage)
|
|
133
|
+
cache_read = usage[:cache_read_tokens] || 0
|
|
134
|
+
cache_write = usage[:cache_write_tokens] || 0
|
|
135
|
+
::RubyLLM::Tokens.new(
|
|
136
|
+
input: [(usage[:input_tokens] || 0) - cache_read - cache_write, 0].max,
|
|
137
|
+
output: usage[:output_tokens] || 0,
|
|
138
|
+
cache_read: cache_read.positive? ? cache_read : nil,
|
|
139
|
+
cache_write: cache_write.positive? ? cache_write : nil,
|
|
140
|
+
# Priced by RubyLLM only for models with a separate reasoning price.
|
|
141
|
+
thinking: usage[:thinking_tokens]
|
|
142
|
+
)
|
|
115
143
|
end
|
|
116
|
-
private_class_method :prices_for
|
|
117
144
|
|
|
118
145
|
# Provider pricing is denominated per 1M tokens; divide here to get
|
|
119
146
|
# the dollar cost for the actual usage count. Named constant for
|
|
120
147
|
# consistency with how RubyLLM and provider docs express prices.
|
|
121
148
|
TOKENS_PER_MILLION = 1_000_000.0
|
|
122
149
|
|
|
150
|
+
# `register_model` validates both prices, so neither is nil here.
|
|
123
151
|
def self.token_cost(tokens, price_per_million)
|
|
124
|
-
(tokens || 0) *
|
|
152
|
+
(tokens || 0) * price_per_million / TOKENS_PER_MILLION
|
|
125
153
|
end
|
|
126
154
|
|
|
127
|
-
def self.find_model(model_name)
|
|
155
|
+
def self.find_model(model_name, provider: nil)
|
|
128
156
|
# Check custom registry first
|
|
129
157
|
custom = @custom_models[model_name]
|
|
130
158
|
return custom if custom
|
|
131
159
|
|
|
132
160
|
return nil unless defined?(RubyLLM)
|
|
133
161
|
|
|
134
|
-
RubyLLM.models.find(model_name)
|
|
162
|
+
provider ? RubyLLM.models.find(model_name, provider: provider) : RubyLLM.models.find(model_name)
|
|
135
163
|
rescue StandardError
|
|
136
164
|
nil
|
|
137
165
|
end
|
|
@@ -146,7 +174,7 @@ module RubyLLM
|
|
|
146
174
|
# to inspect model pricing before invoking `calculate` (e.g., to short-
|
|
147
175
|
# circuit estimate when the model is unknown). Exposing it removes
|
|
148
176
|
# `CostCalculator.send(:find_model)` workarounds at call sites.
|
|
149
|
-
private_class_method :compute_cost, :token_cost, :validate_price!
|
|
177
|
+
private_class_method :compute_cost, :registered_cost, :tokens_for, :token_cost, :validate_price!
|
|
150
178
|
end
|
|
151
179
|
end
|
|
152
180
|
end
|
|
@@ -7,6 +7,9 @@ module RubyLLM
|
|
|
7
7
|
# BaselineDiff#compute_score keys off this prefix to keep skipped cases
|
|
8
8
|
# out of the score denominator. Both ends must read it from here.
|
|
9
9
|
SKIPPED_DETAILS_PREFIX = "skipped:"
|
|
10
|
+
# Not run (no adapter). Readers compare step_status, so any result object
|
|
11
|
+
# exposing it works; kept out of score, pass rate and cost-per-call.
|
|
12
|
+
SKIPPED_STATUS = :skipped
|
|
10
13
|
|
|
11
14
|
def self.skipped_details(reason)
|
|
12
15
|
"#{SKIPPED_DETAILS_PREFIX} #{reason}"
|