ruby_llm-contract 0.10.6 → 1.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.ruby-version +1 -0
- data/CHANGELOG.md +354 -211
- data/README.md +12 -2
- data/docs/guide/getting_started.md +3 -1
- data/docs/guide/llm_judge.md +1 -1
- data/docs/guide/multimodal_input.md +4 -3
- data/docs/guide/output_schema.md +2 -2
- data/docs/guide/testing.md +1 -1
- data/examples/README.md +1 -1
- data/lib/ruby_llm/contract/adapters/ruby_llm.rb +13 -9
- data/lib/ruby_llm/contract/concerns/context_helpers.rb +0 -2
- data/lib/ruby_llm/contract/concerns/deep_symbolize.rb +0 -1
- data/lib/ruby_llm/contract/contract/parser.rb +0 -3
- data/lib/ruby_llm/contract/contract/schema_validator/bound_rule.rb +0 -1
- data/lib/ruby_llm/contract/contract/schema_validator/enum_rule.rb +0 -1
- data/lib/ruby_llm/contract/contract/schema_validator/node.rb +0 -4
- data/lib/ruby_llm/contract/contract/schema_validator/scalar_rules.rb +0 -1
- data/lib/ruby_llm/contract/contract/schema_validator/type_rule.rb +0 -1
- data/lib/ruby_llm/contract/cost_calculator.rb +28 -2
- data/lib/ruby_llm/contract/eval/aggregated_report.rb +4 -0
- data/lib/ruby_llm/contract/eval/baseline_diff.rb +22 -3
- data/lib/ruby_llm/contract/eval/candidate_label.rb +31 -0
- data/lib/ruby_llm/contract/eval/case_executor.rb +2 -2
- data/lib/ruby_llm/contract/eval/case_result.rb +17 -3
- data/lib/ruby_llm/contract/eval/case_result_builder.rb +2 -1
- data/lib/ruby_llm/contract/eval/contract_detail_builder.rb +0 -2
- data/lib/ruby_llm/contract/eval/model_comparison.rb +3 -2
- data/lib/ruby_llm/contract/eval/pipeline_result_adapter.rb +1 -2
- data/lib/ruby_llm/contract/eval/prompt_diff_comparator.rb +0 -1
- data/lib/ruby_llm/contract/eval/prompt_diff_presenter.rb +0 -1
- data/lib/ruby_llm/contract/eval/prompt_diff_serializer.rb +0 -1
- data/lib/ruby_llm/contract/eval/report.rb +1 -1
- data/lib/ruby_llm/contract/eval/report_presenter.rb +0 -1
- data/lib/ruby_llm/contract/eval/report_stats.rb +5 -1
- data/lib/ruby_llm/contract/eval/report_storage.rb +0 -1
- data/lib/ruby_llm/contract/eval/retry_optimizer.rb +10 -20
- data/lib/ruby_llm/contract/eval/step_expectation_applier.rb +2 -1
- data/lib/ruby_llm/contract/eval/trait_evaluator.rb +0 -2
- data/lib/ruby_llm/contract/eval/unknown_cost_gate.rb +40 -0
- data/lib/ruby_llm/contract/eval.rb +2 -0
- data/lib/ruby_llm/contract/minitest.rb +6 -1
- data/lib/ruby_llm/contract/pipeline/runner.rb +0 -1
- data/lib/ruby_llm/contract/pipeline/trace.rb +7 -0
- data/lib/ruby_llm/contract/rake_task/suite_gate.rb +18 -21
- data/lib/ruby_llm/contract/rake_task.rb +9 -4
- data/lib/ruby_llm/contract/rspec/pass_eval.rb +80 -52
- data/lib/ruby_llm/contract/step/base.rb +12 -3
- data/lib/ruby_llm/contract/step/dsl.rb +2 -4
- data/lib/ruby_llm/contract/step/limit_checker.rb +0 -2
- data/lib/ruby_llm/contract/step/retry_executor.rb +0 -2
- data/lib/ruby_llm/contract/step/trace.rb +19 -0
- data/lib/ruby_llm/contract/token_estimator.rb +0 -2
- data/lib/ruby_llm/contract/version.rb +1 -1
- data/lib/ruby_llm/contract.rb +0 -4
- data/ruby_llm-contract.gemspec +6 -5
- metadata +9 -11
- data/.rubocop.yml +0 -58
- data/Gemfile +0 -13
- data/Gemfile.lock +0 -278
- data/Rakefile +0 -8
data/README.md
CHANGED
|
@@ -23,7 +23,7 @@ end
|
|
|
23
23
|
RubyLLM::Contract.configure { }
|
|
24
24
|
```
|
|
25
25
|
|
|
26
|
-
Works with any `ruby_llm` provider (OpenAI, Anthropic, Gemini, etc). Requires `ruby_llm ~>
|
|
26
|
+
Works with any `ruby_llm` provider (OpenAI, Anthropic, Gemini, etc). Requires `ruby_llm ~> 2.0` and Ruby ≥ 3.2.
|
|
27
27
|
|
|
28
28
|
## Example
|
|
29
29
|
|
|
@@ -132,6 +132,8 @@ Everything below is optional — the example above is a complete step. Reach for
|
|
|
132
132
|
|
|
133
133
|
Also supports [multi-step pipelines](docs/guide/pipeline.md) with fail-fast and per-step models.
|
|
134
134
|
|
|
135
|
+
**Runnable companion repo:** [`ruby_llm-contract_demo`](https://github.com/justi/ruby_llm-contract_demo) — full LLM-as-judge lifecycle (v1 strict → v2 drift → judge calibration → v3/v4 iteration) on a returns-policy chatbot scenario, dual-language (EN default, `DEMO_LANG=pl` opt-in). Live OpenAI calls (`gpt-4.1-mini`, `temperature 0`), `LIVE=1` opt-in, ~$0.30 for the full lifecycle. See [llm_judge.md](docs/guide/llm_judge.md) for the methodology.
|
|
136
|
+
|
|
135
137
|
## Relation to `RubyLLM::Agent`
|
|
136
138
|
|
|
137
139
|
`Step::Base` and `RubyLLM::Agent` (since RubyLLM 1.12) are **siblings** targeting the same niche: reusable, class-based prompts. Both call into `RubyLLM::Chat` directly — Step does not wrap Agent. Step adds the contract layer: `validate` (business invariants), `retry_policy escalate(...)` (model escalation on validation failure), `max_cost` pre-flight refusal, regression-eval framework, pipeline composition. **[Full feature mapping →](docs/guide/relation_to_agent.md)**
|
|
@@ -162,10 +164,18 @@ Different layers, complementary. [`ruby_llm-tribunal`](https://github.com/Alqemi
|
|
|
162
164
|
| [Multimodal input (PDF / image / audio)](docs/guide/multimodal_input.md) | Route attachments through the contract; `attachment_token_estimate`, fail-closed cost, calibration table |
|
|
163
165
|
| [Schema DSL reference](docs/guide/output_schema.md) | Every constraint, nested objects, pattern table |
|
|
164
166
|
| [Prompt DSL reference](docs/guide/prompt_ast.md) | `system` / `rule` / `section` / `example` / `user` nodes |
|
|
167
|
+
| [Architecture overview](docs/architecture.md) | Class map: Pipeline, Step, Contract, Prompt, Eval |
|
|
165
168
|
|
|
166
169
|
## Status & versioning
|
|
167
170
|
|
|
168
|
-
|
|
171
|
+
Stable since **1.0.0**. Semver tracked; breaking changes flagged in [CHANGELOG](CHANGELOG.md). Pin `~> 1.0`.
|
|
172
|
+
|
|
173
|
+
What 1.0 promises not to break inside the 1.x line: the documented DSL
|
|
174
|
+
(`prompt`, `output_schema`, `validate`, `retry_policy`, `max_cost`, `define_eval`,
|
|
175
|
+
pipelines), the `Result` and `trace` shapes including `trace[:usage]`'s
|
|
176
|
+
`{ input_tokens:, output_tokens: }` keys, and the adapter interface. A new major
|
|
177
|
+
version of a runtime dependency may require a new major version here — that is
|
|
178
|
+
what 1.0.0 itself was.
|
|
169
179
|
|
|
170
180
|
## FAQ
|
|
171
181
|
|
|
@@ -145,7 +145,7 @@ report.save_baseline!
|
|
|
145
145
|
expect(SummarizeArticle).to pass_eval("regression").without_regressions
|
|
146
146
|
```
|
|
147
147
|
|
|
148
|
-
`without_regressions` fails the build only if a previously-passing case now fails — a new model version, a prompt tweak, or an upstream change that silently lowered quality.
|
|
148
|
+
`without_regressions` fails the build only if a previously-passing case now fails — a new model version, a prompt tweak, or an upstream change that silently lowered quality. A previously-passing case that was skipped this run (no adapter configured) also fails the gate, listed as `SKIPPED` rather than as a regression.
|
|
149
149
|
|
|
150
150
|
## Budget caps
|
|
151
151
|
|
|
@@ -173,6 +173,8 @@ max_cost 0.01, on_unknown_pricing: :warn
|
|
|
173
173
|
|
|
174
174
|
Default is `:refuse`. Use `:warn` only when you accept running without a cost ceiling (fine-tuned models you trust, private endpoints).
|
|
175
175
|
|
|
176
|
+
The eval-level gates - `with_maximum_cost` on `pass_eval`, `maximum_cost` on the rake task, and `maximum_cost:` on `assert_eval_passes` - follow the same rule since 1.1.0. A report's total counts a case it cannot price as $0, so when any case ran on a model without pricing data the gate fails and names the cases instead of comparing an undercounted total. Opt out with `.on_unknown_pricing(:warn)`, `t.on_unknown_pricing = :warn` or `on_unknown_pricing: :warn`, which checks the budget against the priced cases and prints a warning. Offline runs (`sample_response`, a `Test` adapter without `usage:`) report zero tokens, cost $0 with or without pricing, and are never flagged. The check sees only reported tokens: an adapter that returns no token counts looks like a zero-token run.
|
|
177
|
+
|
|
176
178
|
### Preflight cost estimates
|
|
177
179
|
|
|
178
180
|
Check what a call is likely to cost before invoking it:
|
data/docs/guide/llm_judge.md
CHANGED
|
@@ -120,7 +120,7 @@ report = AccuracyJudge.run_eval("calibration")
|
|
|
120
120
|
report.score # => 0.92 → judge agrees with humans 92% of the time
|
|
121
121
|
```
|
|
122
122
|
|
|
123
|
-
A reasonable bar is `score >= 0.85` —
|
|
123
|
+
A reasonable bar is `score >= 0.85` — a practical threshold synthesized from field practice. [Hamel Husain](https://hamel.dev/blog/posts/field-guide/) reports it took **three iterations** of the judge prompt to reach ">90% agreement" with humans in a Honeycomb case study; [Eugene Yan](https://eugeneyan.com/writing/evals/) notes RAG-grounded systems still hold a 5-10% factual inconsistency rate even after good prompt engineering. The specific 0.85 number is the author's summary — neither source quotes it directly. Below ~0.85, judge-human agreement drops into a band where the judge's confident percentages stop matching what a human reviewer would say, and gate decisions become unreliable. If your domain is high-stakes (medical, legal, financial), raise the bar to 0.90+. Iterate the judge's prompt, or escalate to a stronger model, until the score on real production data crosses your bar.
|
|
124
124
|
|
|
125
125
|
**What "iterate the judge's prompt" actually looks like.** The first judge prompt almost always over-flags. Two or three iterations are normal before the judge is ready to gate anything. The common pattern:
|
|
126
126
|
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
> Read this when your contract needs to send a PDF, image, or audio file to the LLM — not just text.
|
|
4
4
|
|
|
5
|
-
`ruby_llm-contract` 0.10.0+ routes attachments through the contract layer, so `max_cost`, `validate`, `retry_policy escalate(...)`, and trace observability still apply. The gem does **not** ship its own multimodal API — it forwards `with:` to `RubyLLM::Chat#ask`, which RubyLLM
|
|
5
|
+
`ruby_llm-contract` 0.10.0+ routes attachments through the contract layer, so `max_cost`, `validate`, `retry_policy escalate(...)`, and trace observability still apply. The gem does **not** ship its own multimodal API — it forwards `with:` to `RubyLLM::Chat#ask`, which RubyLLM normalises per provider (Anthropic, OpenAI, Gemini).
|
|
6
6
|
|
|
7
7
|
## Minimal example
|
|
8
8
|
|
|
@@ -130,7 +130,7 @@ Pick a value at or above the provider's worst-case. The estimate is a **floor fo
|
|
|
130
130
|
|
|
131
131
|
## Multi-turn caveat
|
|
132
132
|
|
|
133
|
-
If your contract uses history (`add_history`), attachments from prior turns are **not** replayed in 0.
|
|
133
|
+
If your contract uses history (`add_history`), attachments from prior turns are **not** replayed (still true in 1.0.x). Single-turn multimodal works; follow-up questions on the same document require additional work that is deferred to a later release. The rationale is recorded in the project's internal ADR-0022, which is not shipped with the gem.
|
|
134
134
|
|
|
135
135
|
## Provider notes
|
|
136
136
|
|
|
@@ -149,7 +149,8 @@ RSpec.describe ExtractInvoiceData do
|
|
|
149
149
|
it "forwards attachment to chat.ask" do
|
|
150
150
|
expect_any_instance_of(RubyLLM::Chat).to receive(:ask)
|
|
151
151
|
.with(anything, with: "fixtures/invoice.pdf")
|
|
152
|
-
.and_return(double(content: '{"vendor":"X",...}',
|
|
152
|
+
.and_return(double(content: '{"vendor":"X",...}',
|
|
153
|
+
tokens: RubyLLM::Tokens.new(input: 200, output: 50)))
|
|
153
154
|
|
|
154
155
|
result = described_class.run("extract", context: { attachment: "fixtures/invoice.pdf" })
|
|
155
156
|
expect(result.status).to eq(:ok)
|
data/docs/guide/output_schema.md
CHANGED
|
@@ -2,12 +2,12 @@
|
|
|
2
2
|
|
|
3
3
|
> Read this as a reference for the schema DSL — every constraint, nested objects, arrays of objects, the full pattern table.
|
|
4
4
|
|
|
5
|
-
Declare the expected output structure using [
|
|
5
|
+
Declare the expected output structure using the [schematist](https://github.com/crmne/schematist) DSL. The schema serves **two purposes**:
|
|
6
6
|
|
|
7
7
|
1. **Output validation** — replaces type and shape checks (enums, ranges, required fields). One declaration instead of many.
|
|
8
8
|
2. **Provider-side request** — with the RubyLLM adapter, the schema is sent to the LLM provider via `chat.with_schema(...)`, asking the model to return JSON matching the shape. Cheaper models sometimes ignore the request, which is why client-side validation (point 1) still matters.
|
|
9
9
|
|
|
10
|
-
> **Same DSL `RubyLLM::Agent.schema` accepts.** Both use `
|
|
10
|
+
> **Same DSL `RubyLLM::Agent.schema` accepts.** Both use `Schematist::Schema.create(&block)` under the hood (schematist is what the `ruby_llm-schema` gem became; RubyLLM 2.x depends on it directly). `Step.output_schema` is eager-compiled at class load and drives client-side validation; `Agent.schema` accepts a `Proc` for dynamic per-call schemas. See [Relation to Agent](relation_to_agent.md) for the full comparison.
|
|
11
11
|
|
|
12
12
|
All examples below extend the `SummarizeArticle` step from the [README](../../README.md).
|
|
13
13
|
|
data/docs/guide/testing.md
CHANGED
|
@@ -160,7 +160,7 @@ end
|
|
|
160
160
|
|
|
161
161
|
- `.with_context(model: "gpt-4.1-mini")` — pick model / pass adapter
|
|
162
162
|
- `.with_minimum_score(0.8)` — gate on average score
|
|
163
|
-
- `.with_maximum_cost(0.01)` — gate on total cost
|
|
163
|
+
- `.with_maximum_cost(0.01)` — gate on total cost; fails when a case's model has no pricing data unless you add `.on_unknown_pricing(:warn)`
|
|
164
164
|
- `.without_regressions` — block any previously-passing case that now fails (reads the baseline)
|
|
165
165
|
- `.compared_with(SummarizeArticleV1)` — A/B against another step; implies regression check
|
|
166
166
|
|
data/examples/README.md
CHANGED
|
@@ -14,7 +14,7 @@ Pedagogical order: hook → activation → evolution → composition → quality
|
|
|
14
14
|
| 05 | `05_eval_dataset.rb` | **"How do I stop silent prompt regressions?"** — define_eval with real cases, baseline vs regressed adapter, regression detection signal, inline eval_case. |
|
|
15
15
|
| 06 | `06_retry_variants.rb` | **"What retry shapes exist beyond cross-model?"** — `attempts: 3` (variance absorption), `reasoning_effort` escalation (low→medium→high), cross-provider fallback (Ollama → Anthropic → OpenAI). |
|
|
16
16
|
|
|
17
|
-
|
|
17
|
+
Four of the seven carry an "Expected output" section in the file header (01, 03, 05, 06), so you can read what they print without running them. The other three differ: `00_basics.rb` prints nothing and annotates each result inline with `# =>`; `02_real_llm_minimal.rb` needs a live provider key, so its output is not deterministic; `04_summarize_and_translate.rb` prints but has no header block.
|
|
18
18
|
|
|
19
19
|
## Running
|
|
20
20
|
|
|
@@ -25,7 +25,6 @@ module RubyLLM
|
|
|
25
25
|
build_response(response)
|
|
26
26
|
end
|
|
27
27
|
|
|
28
|
-
# Maps option keys to the RubyLLM chat method and argument form.
|
|
29
28
|
CHAT_OPTION_METHODS = {
|
|
30
29
|
temperature: :with_temperature,
|
|
31
30
|
schema: :with_schema
|
|
@@ -70,12 +69,10 @@ module RubyLLM
|
|
|
70
69
|
thinking_config = resolve_thinking_config(options)
|
|
71
70
|
chat.with_thinking(**thinking_config) if thinking_config
|
|
72
71
|
|
|
73
|
-
#
|
|
74
|
-
#
|
|
75
|
-
#
|
|
76
|
-
|
|
77
|
-
params[:max_tokens] = options[:max_tokens] if options[:max_tokens]
|
|
78
|
-
chat.with_params(**params) if params.any?
|
|
72
|
+
# RubyLLM 2.0 removed `with_params`; `with_max_output_tokens` is the
|
|
73
|
+
# replacement for this one passthrough. `reasoning_effort` is not
|
|
74
|
+
# forwarded here — it goes through `with_thinking` above.
|
|
75
|
+
chat.with_max_output_tokens(options[:max_tokens]) if options[:max_tokens]
|
|
79
76
|
end
|
|
80
77
|
|
|
81
78
|
# Returns merged `{ effort:, budget: }` or nil. `options[:reasoning_effort]`
|
|
@@ -91,11 +88,18 @@ module RubyLLM
|
|
|
91
88
|
content = response.content
|
|
92
89
|
content = content.to_s unless content.is_a?(Hash) || content.is_a?(Array)
|
|
93
90
|
|
|
91
|
+
# This is the ONLY place upstream token counts are read. RubyLLM 2.0
|
|
92
|
+
# replaced `Message#input_tokens`/`#output_tokens` with a `Tokens` value
|
|
93
|
+
# object. The `{ input_tokens:, output_tokens: }` shape below is this
|
|
94
|
+
# gem's own public contract (Step::Trace#usage, cost calculation,
|
|
95
|
+
# documented in the README) and deliberately does NOT change.
|
|
96
|
+
tokens = response.tokens
|
|
97
|
+
|
|
94
98
|
Response.new(
|
|
95
99
|
content: content,
|
|
96
100
|
usage: {
|
|
97
|
-
input_tokens:
|
|
98
|
-
output_tokens:
|
|
101
|
+
input_tokens: tokens&.input || 0,
|
|
102
|
+
output_tokens: tokens&.output || 0
|
|
99
103
|
}
|
|
100
104
|
)
|
|
101
105
|
end
|
|
@@ -38,7 +38,6 @@ module RubyLLM
|
|
|
38
38
|
end
|
|
39
39
|
private_class_method :parse_json_text
|
|
40
40
|
|
|
41
|
-
# Fallback: attempt to extract the first JSON object or array from prose
|
|
42
41
|
def self.parse_json_with_extraction(text, raw_output)
|
|
43
42
|
extracted = extract_json(text)
|
|
44
43
|
unless extracted
|
|
@@ -72,8 +71,6 @@ module RubyLLM
|
|
|
72
71
|
match ? match[1] : text
|
|
73
72
|
end
|
|
74
73
|
|
|
75
|
-
# Extract the first JSON object or array from text that may contain prose.
|
|
76
|
-
# Uses bracket-matching to find the outermost balanced { } or [ ] block.
|
|
77
74
|
JSON_START_PATTERN = /[{\[]/
|
|
78
75
|
|
|
79
76
|
def self.extract_json(text)
|
|
@@ -84,11 +84,37 @@ module RubyLLM
|
|
|
84
84
|
end
|
|
85
85
|
|
|
86
86
|
def self.compute_cost(model_info, usage)
|
|
87
|
-
|
|
88
|
-
|
|
87
|
+
prices = prices_for(model_info)
|
|
88
|
+
return nil unless prices
|
|
89
|
+
|
|
90
|
+
input_cost = token_cost(usage[:input_tokens], prices[:input])
|
|
91
|
+
output_cost = token_cost(usage[:output_tokens], prices[:output])
|
|
89
92
|
(input_cost + output_cost).round(6)
|
|
90
93
|
end
|
|
91
94
|
|
|
95
|
+
# Two shapes reach here. `RegisteredModel` (our own struct, from
|
|
96
|
+
# `register_model`) exposes flat `*_price_per_million` readers. RubyLLM 2.0
|
|
97
|
+
# moved provider pricing into a nested value object and dropped those flat
|
|
98
|
+
# readers, so it has to be walked: pricing -> text_tokens -> standard.
|
|
99
|
+
#
|
|
100
|
+
# nil means "unknown pricing". The guards keep the two known shapes
|
|
101
|
+
# explicit, but `calculate` rescues StandardError, so a future shape this
|
|
102
|
+
# walk cannot read also degrades to nil: `max_cost` then refuses the call
|
|
103
|
+
# citing "no pricing data" rather than the real cause, and paths that floor
|
|
104
|
+
# unknown cost to 0.0 (eval cost estimates, report totals) undercount.
|
|
105
|
+
def self.prices_for(model_info)
|
|
106
|
+
if model_info.respond_to?(:input_price_per_million)
|
|
107
|
+
return { input: model_info.input_price_per_million,
|
|
108
|
+
output: model_info.output_price_per_million }
|
|
109
|
+
end
|
|
110
|
+
|
|
111
|
+
tier = model_info.pricing&.text_tokens&.standard if model_info.respond_to?(:pricing)
|
|
112
|
+
return unless tier.respond_to?(:input_per_million)
|
|
113
|
+
|
|
114
|
+
{ input: tier.input_per_million, output: tier.output_per_million }
|
|
115
|
+
end
|
|
116
|
+
private_class_method :prices_for
|
|
117
|
+
|
|
92
118
|
# Provider pricing is denominated per 1M tokens; divide here to get
|
|
93
119
|
# the dollar cost for the actual usage count. Named constant for
|
|
94
120
|
# consistency with how RubyLLM and provider docs express prices.
|
|
@@ -44,6 +44,10 @@ module RubyLLM
|
|
|
44
44
|
@runs.sum(&:total_cost) / @runs.length.to_f
|
|
45
45
|
end
|
|
46
46
|
|
|
47
|
+
def unknown_cost_results
|
|
48
|
+
@runs.flat_map(&:unknown_cost_results)
|
|
49
|
+
end
|
|
50
|
+
|
|
47
51
|
def avg_latency_ms
|
|
48
52
|
latencies = @runs.filter_map(&:avg_latency_ms)
|
|
49
53
|
return nil if latencies.empty?
|
|
@@ -19,6 +19,7 @@ module RubyLLM
|
|
|
19
19
|
current = @current[name]
|
|
20
20
|
next unless current
|
|
21
21
|
next unless baseline[:passed] && !current[:passed]
|
|
22
|
+
next if skipped?(current)
|
|
22
23
|
|
|
23
24
|
{
|
|
24
25
|
case: name,
|
|
@@ -47,8 +48,19 @@ module RubyLLM
|
|
|
47
48
|
(current_score - baseline_score).round(4)
|
|
48
49
|
end
|
|
49
50
|
|
|
51
|
+
# Passed in the baseline, not run now (no adapter): not evidence of a
|
|
52
|
+
# regression, but not evidence it still passes either, so the gate fails.
|
|
53
|
+
def skipped_passing_cases
|
|
54
|
+
@baseline.filter_map do |name, baseline|
|
|
55
|
+
current = @current[name]
|
|
56
|
+
next unless current && baseline[:passed] && skipped?(current)
|
|
57
|
+
|
|
58
|
+
{ case: name, detail: current[:details] }
|
|
59
|
+
end
|
|
60
|
+
end
|
|
61
|
+
|
|
50
62
|
def regressed?
|
|
51
|
-
regressions.any? || removed_passing_cases.any?
|
|
63
|
+
regressions.any? || removed_passing_cases.any? || skipped_passing_cases.any?
|
|
52
64
|
end
|
|
53
65
|
|
|
54
66
|
def removed_passing_cases
|
|
@@ -70,6 +82,7 @@ module RubyLLM
|
|
|
70
82
|
def to_s
|
|
71
83
|
lines = ["Score: #{baseline_score.round(2)} → #{current_score.round(2)} (#{format_delta})"]
|
|
72
84
|
regressions.each { |r| lines << " REGRESSED #{r[:case]}: #{r[:detail]}" }
|
|
85
|
+
skipped_passing_cases.each { |s| lines << " SKIPPED #{s[:case]}: #{s[:detail]}" }
|
|
73
86
|
improvements.each { |r| lines << " IMPROVED #{r[:case]}" }
|
|
74
87
|
new_cases.each { |c| lines << " NEW #{c}" }
|
|
75
88
|
removed_cases.each { |c| lines << " REMOVED #{c}" }
|
|
@@ -79,13 +92,19 @@ module RubyLLM
|
|
|
79
92
|
private
|
|
80
93
|
|
|
81
94
|
def compute_score(cases)
|
|
82
|
-
#
|
|
83
|
-
|
|
95
|
+
# Asymmetric on purpose: the baseline side arrives already filtered by
|
|
96
|
+
# ReportStats#evaluated_results, the current side does not, so this
|
|
97
|
+
# string check is the only skip filter on one of the two sides.
|
|
98
|
+
evaluated = cases.reject { |c| skipped?(c) }
|
|
84
99
|
return 0.0 if evaluated.empty?
|
|
85
100
|
|
|
86
101
|
evaluated.sum { |c| c[:score] } / evaluated.length
|
|
87
102
|
end
|
|
88
103
|
|
|
104
|
+
def skipped?(serialized_case)
|
|
105
|
+
serialized_case[:details]&.start_with?(CaseResult::SKIPPED_DETAILS_PREFIX)
|
|
106
|
+
end
|
|
107
|
+
|
|
89
108
|
def index_by_name(cases)
|
|
90
109
|
cases.each_with_object({}) { |c, h| h[c[:name]] = c }
|
|
91
110
|
end
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module RubyLLM
|
|
4
|
+
module Contract
|
|
5
|
+
module Eval
|
|
6
|
+
# The `model (effort: x)` notation, rendered into optimizer tables and parsed
|
|
7
|
+
# back to rebuild candidate configs. Render and parse live together because a
|
|
8
|
+
# format change on one side silently mis-parses on the other: the pattern is
|
|
9
|
+
# built from the same literal the renderer emits.
|
|
10
|
+
module CandidateLabel
|
|
11
|
+
OPEN = " (effort: "
|
|
12
|
+
CLOSE = ")"
|
|
13
|
+
PATTERN = /\s*#{Regexp.escape(OPEN.lstrip)}(\w+)#{Regexp.escape(CLOSE)}/
|
|
14
|
+
|
|
15
|
+
def self.render(config)
|
|
16
|
+
effort = config[:reasoning_effort]
|
|
17
|
+
return config[:model] unless effort
|
|
18
|
+
|
|
19
|
+
"#{config[:model]}#{OPEN}#{effort}#{CLOSE}"
|
|
20
|
+
end
|
|
21
|
+
|
|
22
|
+
def self.parse(label)
|
|
23
|
+
match = PATTERN.match(label)
|
|
24
|
+
return { model: label } unless match
|
|
25
|
+
|
|
26
|
+
{ model: match.pre_match.strip, reasoning_effort: match[1] }
|
|
27
|
+
end
|
|
28
|
+
end
|
|
29
|
+
end
|
|
30
|
+
end
|
|
31
|
+
end
|
|
@@ -30,7 +30,7 @@ module RubyLLM
|
|
|
30
30
|
private
|
|
31
31
|
|
|
32
32
|
def missing_adapter?(error)
|
|
33
|
-
error.message.include?(
|
|
33
|
+
error.message.include?(Step::Base::NO_ADAPTER_MESSAGE)
|
|
34
34
|
end
|
|
35
35
|
|
|
36
36
|
def skipped_result(test_case, reason)
|
|
@@ -43,7 +43,7 @@ module RubyLLM
|
|
|
43
43
|
score: 0.0,
|
|
44
44
|
passed: false,
|
|
45
45
|
label: "SKIP",
|
|
46
|
-
details:
|
|
46
|
+
details: CaseResult.skipped_details(reason)
|
|
47
47
|
)
|
|
48
48
|
end
|
|
49
49
|
end
|
|
@@ -4,11 +4,20 @@ module RubyLLM
|
|
|
4
4
|
module Contract
|
|
5
5
|
module Eval
|
|
6
6
|
class CaseResult
|
|
7
|
+
# BaselineDiff#compute_score keys off this prefix to keep skipped cases
|
|
8
|
+
# out of the score denominator. Both ends must read it from here.
|
|
9
|
+
SKIPPED_DETAILS_PREFIX = "skipped:"
|
|
10
|
+
|
|
11
|
+
def self.skipped_details(reason)
|
|
12
|
+
"#{SKIPPED_DETAILS_PREFIX} #{reason}"
|
|
13
|
+
end
|
|
14
|
+
|
|
7
15
|
attr_reader :name, :input, :output, :expected, :step_status,
|
|
8
16
|
:score, :details, :duration_ms, :cost, :attempts
|
|
9
17
|
|
|
10
|
-
def initialize(name:, input:, output:, expected:, step_status:,
|
|
11
|
-
score:, passed:, label: nil, details: nil, duration_ms: nil, cost: nil, attempts: nil
|
|
18
|
+
def initialize(name:, input:, output:, expected:, step_status:, # rubocop:disable Metrics/ParameterLists
|
|
19
|
+
score:, passed:, label: nil, details: nil, duration_ms: nil, cost: nil, attempts: nil,
|
|
20
|
+
cost_unknown: false)
|
|
12
21
|
@name = name
|
|
13
22
|
@input = input
|
|
14
23
|
@output = output
|
|
@@ -21,6 +30,7 @@ module RubyLLM
|
|
|
21
30
|
@duration_ms = duration_ms
|
|
22
31
|
@cost = cost
|
|
23
32
|
@attempts = attempts
|
|
33
|
+
@cost_unknown = cost_unknown
|
|
24
34
|
freeze
|
|
25
35
|
end
|
|
26
36
|
|
|
@@ -32,6 +42,10 @@ module RubyLLM
|
|
|
32
42
|
!@passed
|
|
33
43
|
end
|
|
34
44
|
|
|
45
|
+
def cost_unknown?
|
|
46
|
+
@cost_unknown
|
|
47
|
+
end
|
|
48
|
+
|
|
35
49
|
def label
|
|
36
50
|
@label || (@passed ? "PASS" : "FAIL")
|
|
37
51
|
end
|
|
@@ -61,7 +75,7 @@ module RubyLLM
|
|
|
61
75
|
duration_ms: @duration_ms,
|
|
62
76
|
cost: @cost,
|
|
63
77
|
attempts: @attempts
|
|
64
|
-
}
|
|
78
|
+
}.tap { |hash| hash[:cost_unknown] = true if @cost_unknown }
|
|
65
79
|
end
|
|
66
80
|
|
|
67
81
|
private
|
|
@@ -19,7 +19,8 @@ module RubyLLM
|
|
|
19
19
|
details: evaluation.details,
|
|
20
20
|
duration_ms: trace_metric(trace, :total_latency_ms, :latency_ms),
|
|
21
21
|
cost: trace_metric(trace, :total_cost, :cost),
|
|
22
|
-
attempts: trace_attempts(trace)
|
|
22
|
+
attempts: trace_attempts(trace),
|
|
23
|
+
cost_unknown: trace.respond_to?(:cost_unknown?) && trace.cost_unknown?
|
|
23
24
|
)
|
|
24
25
|
end
|
|
25
26
|
|
|
@@ -6,9 +6,10 @@ module RubyLLM
|
|
|
6
6
|
class ModelComparison
|
|
7
7
|
attr_reader :eval_name, :reports, :configs, :fallback
|
|
8
8
|
|
|
9
|
+
# Kept as the name six call sites already use; the format itself lives in
|
|
10
|
+
# CandidateLabel, which owns both directions.
|
|
9
11
|
def self.candidate_label(config)
|
|
10
|
-
|
|
11
|
-
effort ? "#{config[:model]} (effort: #{effort})" : config[:model]
|
|
12
|
+
CandidateLabel.render(config)
|
|
12
13
|
end
|
|
13
14
|
|
|
14
15
|
def initialize(eval_name:, reports:, configs: nil, fallback: nil)
|
|
@@ -3,8 +3,7 @@
|
|
|
3
3
|
module RubyLLM
|
|
4
4
|
module Contract
|
|
5
5
|
module Eval
|
|
6
|
-
#
|
|
7
|
-
# Replaces OpenStruct usage in Runner#normalize_pipeline_result.
|
|
6
|
+
# Replaces OpenStruct usage in StepResultNormalizer#normalize_pipeline_result.
|
|
8
7
|
PipelineResultAdapter = Struct.new(:status, :ok_flag, :parsed_output, :validation_errors, :trace) do
|
|
9
8
|
def ok?
|
|
10
9
|
ok_flag
|
|
@@ -15,7 +15,7 @@ module RubyLLM
|
|
|
15
15
|
BASELINE_DIR = ".eval_baselines"
|
|
16
16
|
|
|
17
17
|
def_delegators :@stats, :score, :passed, :failed, :skipped, :failures, :pass_rate, :pass_rate_ratio,
|
|
18
|
-
:total_cost, :avg_latency_ms, :passed?,
|
|
18
|
+
:total_cost, :unknown_cost_results, :avg_latency_ms, :passed?,
|
|
19
19
|
:production_mode?, :escalation_rate, :single_shot_cost, :single_shot_latency_ms,
|
|
20
20
|
:effective_cost, :effective_latency_ms, :latency_percentiles
|
|
21
21
|
def_delegators :@presenter, :summary, :to_s, :print_summary
|
|
@@ -3,7 +3,6 @@
|
|
|
3
3
|
module RubyLLM
|
|
4
4
|
module Contract
|
|
5
5
|
module Eval
|
|
6
|
-
# Computes aggregate metrics for an eval report.
|
|
7
6
|
class ReportStats
|
|
8
7
|
def initialize(results:)
|
|
9
8
|
@results = results
|
|
@@ -45,6 +44,11 @@ module RubyLLM
|
|
|
45
44
|
@results.sum { |result| result.cost || 0.0 }
|
|
46
45
|
end
|
|
47
46
|
|
|
47
|
+
# Cases total_cost could not price; it counts them as 0.0.
|
|
48
|
+
def unknown_cost_results
|
|
49
|
+
evaluated_results.select(&:cost_unknown?)
|
|
50
|
+
end
|
|
51
|
+
|
|
48
52
|
def avg_latency_ms
|
|
49
53
|
latencies = @results.filter_map(&:duration_ms)
|
|
50
54
|
return nil if latencies.empty?
|