ruby_llm-contract 0.10.5 → 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +124 -5
- data/README.md +12 -2
- data/docs/guide/llm_judge.md +3 -3
- data/docs/guide/multimodal_input.md +4 -3
- data/docs/guide/output_schema.md +2 -2
- data/docs/guide/relation_to_tribunal.md +28 -0
- data/examples/README.md +1 -1
- data/lib/ruby_llm/contract/adapters/ruby_llm.rb +13 -9
- data/lib/ruby_llm/contract/concerns/context_helpers.rb +0 -2
- data/lib/ruby_llm/contract/concerns/deep_symbolize.rb +0 -1
- data/lib/ruby_llm/contract/contract/parser.rb +0 -3
- data/lib/ruby_llm/contract/contract/schema_validator/bound_rule.rb +0 -1
- data/lib/ruby_llm/contract/contract/schema_validator/enum_rule.rb +0 -1
- data/lib/ruby_llm/contract/contract/schema_validator/node.rb +0 -4
- data/lib/ruby_llm/contract/contract/schema_validator/scalar_rules.rb +0 -1
- data/lib/ruby_llm/contract/contract/schema_validator/type_rule.rb +0 -1
- data/lib/ruby_llm/contract/cost_calculator.rb +27 -2
- data/lib/ruby_llm/contract/eval/baseline_diff.rb +4 -2
- data/lib/ruby_llm/contract/eval/candidate_label.rb +31 -0
- data/lib/ruby_llm/contract/eval/case_executor.rb +2 -2
- data/lib/ruby_llm/contract/eval/case_result.rb +8 -0
- data/lib/ruby_llm/contract/eval/contract_detail_builder.rb +0 -2
- data/lib/ruby_llm/contract/eval/model_comparison.rb +3 -2
- data/lib/ruby_llm/contract/eval/pipeline_result_adapter.rb +1 -2
- data/lib/ruby_llm/contract/eval/prompt_diff_comparator.rb +0 -1
- data/lib/ruby_llm/contract/eval/prompt_diff_presenter.rb +0 -1
- data/lib/ruby_llm/contract/eval/prompt_diff_serializer.rb +0 -1
- data/lib/ruby_llm/contract/eval/report_presenter.rb +0 -1
- data/lib/ruby_llm/contract/eval/report_stats.rb +0 -1
- data/lib/ruby_llm/contract/eval/report_storage.rb +0 -1
- data/lib/ruby_llm/contract/eval/retry_optimizer.rb +10 -20
- data/lib/ruby_llm/contract/eval/trait_evaluator.rb +0 -2
- data/lib/ruby_llm/contract/eval.rb +1 -0
- data/lib/ruby_llm/contract/pipeline/runner.rb +0 -1
- data/lib/ruby_llm/contract/rake_task/suite_gate.rb +0 -12
- data/lib/ruby_llm/contract/rake_task.rb +1 -2
- data/lib/ruby_llm/contract/step/base.rb +12 -3
- data/lib/ruby_llm/contract/step/dsl.rb +2 -4
- data/lib/ruby_llm/contract/step/limit_checker.rb +0 -2
- data/lib/ruby_llm/contract/step/retry_executor.rb +0 -2
- data/lib/ruby_llm/contract/token_estimator.rb +0 -2
- data/lib/ruby_llm/contract/version.rb +1 -1
- data/lib/ruby_llm/contract.rb +0 -4
- data/ruby_llm-contract.gemspec +6 -5
- metadata +7 -11
- data/.rubocop.yml +0 -58
- data/Gemfile +0 -13
- data/Gemfile.lock +0 -278
- data/Rakefile +0 -8
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: d7d18fbcf79a634c5f577d8965bc82ed10c12a8593b244ad3f07c28f21033035
|
|
4
|
+
data.tar.gz: '009718225fb17d5b34ea3fe3bf290b812fb17321faa1278d027ae8bdc66e9c2c'
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: 2db5376a4aefe02acd3ce2a2debc645d2c2b14d185877da6cbccd9810b04aea8f1a3c830d48d693a66ae8268774ff5041ee81d3c3eba2e33cdc22e5d0ec15d0e
|
|
7
|
+
data.tar.gz: f560b145abc0b5114240edec2cc01fe563b9d51f48ef90c80a72e9b6b121e3246b7c9ac3196027471bd5450b3b3225f53f566d2d15c50976b8106fda889a458a
|
data/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,124 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 1.0.0 (2026-09-27)
|
|
4
|
+
|
|
5
|
+
**Requires `ruby_llm ~> 2.0`.** This is the only breaking change: it is a
|
|
6
|
+
dependency requirement, not a change to this gem's API. Your call sites,
|
|
7
|
+
`output_schema` blocks, `trace[:usage]` keys and adapter interface are all
|
|
8
|
+
untouched — if you are on ruby_llm 2.x, upgrading should need no code edits.
|
|
9
|
+
|
|
10
|
+
1.0 also states what is now stable: the documented DSL, the `Result`/`trace`
|
|
11
|
+
shapes including `trace[:usage]`'s `{ input_tokens:, output_tokens: }` keys, and
|
|
12
|
+
the adapter interface. A future major version of a runtime dependency may again
|
|
13
|
+
require a major version here — that is what this release is.
|
|
14
|
+
|
|
15
|
+
### Added
|
|
16
|
+
|
|
17
|
+
- **CI, for the first time.** Matrix over Ruby 3.2 and 3.4 against both
|
|
18
|
+
ruby_llm 2.0.0 (pinned, so a regression at the major boundary stays
|
|
19
|
+
distinguishable) and the newest 2.x the gemspec allows. The job also runs every
|
|
20
|
+
offline example, builds the gem and fails on any `gem build` warning, and
|
|
21
|
+
installs the built gem to prove it `require`s. The seven live-key specs are
|
|
22
|
+
bounded explicitly: CI asserts the pending count is exactly 7 and that all
|
|
23
|
+
seven come from `spec/integration/cost_of_quality_real_spec.rb`, so a failing
|
|
24
|
+
spec cannot be quietly downgraded to pending.
|
|
25
|
+
- **`.rubocop_todo.yml`** so lint can be enforced in CI at zero offences while the
|
|
26
|
+
pre-existing backlog stays visible and shrinkable, rather than blocking the gate.
|
|
27
|
+
- **A boundary spec for the RubyLLM 2.x seam**
|
|
28
|
+
(`spec/ruby_llm/contract/adapters/ruby_llm_2x_boundary_spec.rb`). Every
|
|
29
|
+
upstream-version-specific detail is supposed to live in `Adapters::RubyLLM` and
|
|
30
|
+
`CostCalculator`; these tests pin that, including that `max_cost` still fails
|
|
31
|
+
closed if provider pricing ever becomes unreachable again.
|
|
32
|
+
|
|
33
|
+
### Changed
|
|
34
|
+
|
|
35
|
+
- **`ruby_llm` requirement raised to `~> 2.0`** (from `~> 1.12`).
|
|
36
|
+
- **`ruby_llm-schema` replaced by `schematist ~> 1.1`.** `ruby_llm-schema` 1.0.0 is
|
|
37
|
+
31 lines that emit a deprecation warning and alias `RubyLLM::Schema =
|
|
38
|
+
Schematist::Schema`; ruby_llm 2.x depends on schematist directly. The
|
|
39
|
+
`output_schema do ... end` DSL is unchanged — same object behind a different
|
|
40
|
+
constant.
|
|
41
|
+
- **Adapter reads token counts from `Message#tokens`.** RubyLLM 2.0 replaced
|
|
42
|
+
`Message#input_tokens`/`#output_tokens` with a `Tokens` value object. This is the
|
|
43
|
+
only place in the gem that touches upstream token counts, and the normalised
|
|
44
|
+
`{ input_tokens:, output_tokens: }` hash it publishes is deliberately unchanged.
|
|
45
|
+
- **`max_tokens` now forwards via `with_max_output_tokens`**; `Chat#with_params` no
|
|
46
|
+
longer exists in 2.0.
|
|
47
|
+
|
|
48
|
+
### Fixed
|
|
49
|
+
|
|
50
|
+
- **`estimate_eval_cost` raised `TypeError` instead of flooring at $0.00.** With
|
|
51
|
+
ruby_llm 2.x, `CostCalculator#calculate` returns nil for two distinct misses -
|
|
52
|
+
model absent from the registry, and model present but with unreadable pricing -
|
|
53
|
+
and the summation guarded only the first. Measured against the real 2.0.0
|
|
54
|
+
registry, 669 of 1672 models take the second path, so any eval estimate touching
|
|
55
|
+
one of them crashed. `getting_started.md` documents this method as a floor, not
|
|
56
|
+
a fail-closed, and it behaves that way again.
|
|
57
|
+
- **The optimizer's candidate table could mangle a model name containing
|
|
58
|
+
brackets.** The short label was built by cutting the rendered label apart, so
|
|
59
|
+
`"custom (beta) (effort: high)"` came out as `"custom (beta@high)"` - the
|
|
60
|
+
model's own closing bracket was consumed. The notation now has one owner
|
|
61
|
+
(`Eval::CandidateLabel`, which both renders and parses), and the short form is
|
|
62
|
+
composed from parsed parts instead of string surgery.
|
|
63
|
+
- **A reworded skipped-case reason could have reported a phantom regression.**
|
|
64
|
+
`BaselineDiff` excludes skipped cases from the score denominator by matching a
|
|
65
|
+
string prefix that `CaseExecutor` writes, and the two sides are filtered
|
|
66
|
+
differently - the baseline side structurally, the current side only by that
|
|
67
|
+
string. Both ends now read one constant, and a baseline file written by 0.10.6
|
|
68
|
+
is covered by its own spec (the fixture was generated by installing 0.10.6, not
|
|
69
|
+
written by hand).
|
|
70
|
+
- **Cost calculation against ruby_llm 2.x.** 2.0 dropped the flat
|
|
71
|
+
`input_price_per_million` / `output_price_per_million` readers on `Model` for a
|
|
72
|
+
nested `pricing -> text_tokens -> standard` walk. Because `CostCalculator`
|
|
73
|
+
rescued `StandardError` into a nil cost, every priced model silently became
|
|
74
|
+
"unknown pricing" — and since `max_cost` fails closed on unknown pricing, every
|
|
75
|
+
budgeted call would have been refused. Pricing extraction is now explicit for
|
|
76
|
+
both shapes (upstream models and locally `register_model`-ed ones), and an
|
|
77
|
+
unrecognised shape returns nil rather than a swallowed exception.
|
|
78
|
+
- **A shipped doc example taught the wrong test double.**
|
|
79
|
+
`docs/guide/multimodal_input.md` showed `double(content:, input_tokens:,
|
|
80
|
+
output_tokens:)`, which a verified double rejects against ruby_llm 2.0.
|
|
81
|
+
- **`FINDING 7` in the audit specs asserted a constraint on `ruby_llm-schema`**, a
|
|
82
|
+
dependency this release removes. Generalised to "no runtime dependency is
|
|
83
|
+
declared without a version constraint", so the check cannot be retired by
|
|
84
|
+
renaming the gem it happened to be about.
|
|
85
|
+
- Stale version and dependency claims in `README.md`, `docs/guide/output_schema.md`
|
|
86
|
+
and `docs/guide/multimodal_input.md`.
|
|
87
|
+
|
|
88
|
+
### Housekeeping (was staged as 0.11.0)
|
|
89
|
+
|
|
90
|
+
Result of a 20-pass systematic audit of the whole repository. No API changes and no behaviour change inside `Step`/`Pipeline`/`Eval`; the user-visible differences are in what the gem **packages**, what the RubyGems page **links**, and what the docs **claim**.
|
|
91
|
+
|
|
92
|
+
### Changed
|
|
93
|
+
|
|
94
|
+
- **The published gem no longer ships development-only files.** Measured by building the gem and unpacking it: `.rubocop.yml`, `Gemfile`, `Gemfile.lock` and `Rakefile` were being packaged, contradicting the gemspec's own stated policy ("dev configs excluded so the published gem contains only what adopters actually need at runtime"). The shipped `Rakefile` was unusable anyway — it `require`s rspec, which is not a runtime dependency, and defines a task over `spec/`, which the same build excludes. Package contents: **137 → 133 files**; `lib/` (105) and `docs/` (16) unchanged.
|
|
95
|
+
- **`Gemfile.lock` is no longer tracked.** It was in the index despite being matched by `.gitignore:10`, and was repacked on every release. A library should not pin its adopters' dependency versions.
|
|
96
|
+
- **The RubyGems page will now show a "Source Code" link.** `metadata["homepage_uri"]` duplicated `spec.homepage` and, being listed first, suppressed `source_code_uri` in the UI — `gem build` warned about this on every build. Dropped the duplicate; `source_code_uri` stays, because `bundle info` and the RubyGems API read it as a distinct field.
|
|
97
|
+
|
|
98
|
+
### Fixed
|
|
99
|
+
|
|
100
|
+
- **`.rubocop.yml` declared `AllCops` twice.** Psych applies last-key-wins, so `TargetRubyVersion: 3.2`, `NewCops: enable` and `SuggestExtensions: false` were silently discarded, and the second block's `Exclude` **replaced** RuboCop's defaults instead of adding to them. Effect: the linter scanned 353 files / 2423 offences, 147 of those files untracked under `tmp/` (they carried 2155 of the offences). Merging the blocks alone does not fix it — `inherit_mode: merge` is what restores the defaults. Now 206 files / 300 offences, all pre-existing formatting on real code. (Both counts measured with rubocop 1.88.0; `Gemfile.lock` is no longer tracked, so pin the tool when reproducing.)
|
|
101
|
+
- **`docs/guide/multimodal_input.md` linked to a file that could never resolve** — `../decisions/ADR-0022-*.md` points at `docs/decisions/` (no such directory; the ADR lives in `doc/decisions/`, singular) and that path is gitignored. Since `docs/guide/*` is packaged, this reproduced exactly the "links 404'd locally" defect that 0.10.2 set out to fix. Same dangling reference removed from the 0.9.0 CHANGELOG entry.
|
|
102
|
+
- **Two documented rake-task environment variables never existed.** `REASONING_EFFORT=` and `OLLAMA_API_BASE=` were listed in the 0.5.0 CHANGELOG entry, but `git log --all -S<name> -- lib/` is empty for both — they were never read at any point in the project's history, and ruby_llm's own configuration only seeds `RUBYLLM_DEBUG` / `RUBYLLM_STREAM_DEBUG` from the environment. Setting either was a silent no-op. Reasoning effort is selected with `CANDIDATES=model@effort`; Ollama's base URL goes through `RubyLLM.configure`.
|
|
103
|
+
- **README stated the wrong current version** (0.10.4 while `version.rb` was 0.10.6) — the one sentence that has to change every release.
|
|
104
|
+
- **`examples/README.md` claimed every example carries an "Expected output" section**; four of seven do, for three different reasons, so the sentence was the error rather than the files.
|
|
105
|
+
- **`docs/architecture.md` was reachable from nowhere** — the only one of 16 tracked docs with no inbound link. Added to the README index.
|
|
106
|
+
|
|
107
|
+
### Removed
|
|
108
|
+
|
|
109
|
+
- Dead code with a full evidence battery behind each cut: `SchemaValidator::Node#numeric?` (zero readers; the one place needing the check inlines `value.is_a?(Numeric)`), an abandoned twin test helper whose sibling has 14 call sites, three unused `let`/helper definitions, four useless assignments, seven redundant `require`s in specs, and a `SimpleCov.start` that was a measured no-op because `require "simplecov"` already autoloads `.simplecov`.
|
|
110
|
+
- 31 WHAT comment blocks (51 lines) in `lib/` plus 20 comment lines in `spec/`, all restating the line or class below them - including all six `Extracted from ...` comments, five of them the "to reduce class length" boilerplate (refactor history git already records) and ten one-line class docstrings paraphrasing their own class name. WHY comments — invariants, external-behaviour notes, anti-duplication markers — were deliberately kept; so were the three evaluator banners that name a comparison semantic the class name cannot convey.
|
|
111
|
+
- Private-project residue: a spec block still labelled "reddit promo planner shape" although that project was removed as private in #25, and four dangling `(Batch N / TODO)` references to a `TODO.md` deleted in 0.10.1.
|
|
112
|
+
|
|
113
|
+
## 0.10.6 (2026-06-11)
|
|
114
|
+
|
|
115
|
+
Docs accuracy patch: correct the positioning of `llm_judge.md` against `ruby_llm-tribunal` after a deeper audit of Tribunal's documented scope. The previous wording ("you are reinventing what `ruby_llm-tribunal` ships as a built-in catalog") read as if Tribunal made `llm_judge.md` redundant — incorrect. Tribunal's README ships an off-the-shelf implementation catalog (`assert_faithful`, `assert_hallucination`, `assert_refusal`, `assert_no_pii`, etc.) but does **not** document calibration workflow, per-claim breakdown, judge-prompt iteration, or judge-as-`evaluator:` integration — exactly the methodology `llm_judge.md` covers. The two are complementary layers, not alternatives. No code behaviour change.
|
|
116
|
+
|
|
117
|
+
### Fixed
|
|
118
|
+
|
|
119
|
+
- **`docs/guide/relation_to_tribunal.md`** — added the **"Tribunal's catalog vs Contract's `llm_judge.md` — concrete decision tree"** sub-section under "When to use which", giving adopters a sharp three-way fork: reach for Tribunal's catalog when the check is domain-general (faithfulness vs context, hallucination, refusal, PII, jailbreak, toxicity, bias); build a custom judge per `llm_judge.md` when the criterion is domain-specific, when the judge needs to live inside a `define_eval` regression gate as the `evaluator:` lambda, when per-claim sentence-level debug output is required, or when the judge prompt itself needs to be iterated against your data; use both for the same project at different lifecycle stages (Tribunal at spec-time, calibrated custom judge at CI merge gate). Added the **"What Tribunal documents — and what it doesn't"** sub-section with a five-row comparison table making explicit which methodology gaps `llm_judge.md` covers that Tribunal's README leaves to the adopter (calibration against humans, prompt iteration on over-flag, per-claim breakdown, evaluator-lambda integration, the six anti-patterns).
|
|
120
|
+
- **`docs/guide/llm_judge.md`** — rewrote the closing "When to escalate to Tribunal's catalog" section as **"When to reach for Tribunal instead"**: shorter, accurate (Tribunal is a complementary catalog, not a replacement), points to `relation_to_tribunal.md` for the full decision tree and integration patterns. The previous wording implied that building any of the four standard judges (faithful / hallucination / refusal / PII) was "reinvention" — true for the **implementation** (Tribunal ships them), false for the **methodology** (Tribunal's README doesn't document calibration, anti-patterns, or per-claim breakdown). The methodology applies equally to Tribunal's built-ins, Tribunal's custom registered judges, and Contract `Step::Base` judges.
|
|
121
|
+
|
|
3
122
|
## 0.10.5 (2026-06-11)
|
|
4
123
|
|
|
5
124
|
Docs release: new `llm_judge.md` guide + comprehensive clarity audit across all 16 shipping documentation files. No code behaviour change.
|
|
@@ -94,7 +213,7 @@ validate("score in range 0-100") { |o| o[:score].between?(0, 100) }
|
|
|
94
213
|
|
|
95
214
|
### Added
|
|
96
215
|
|
|
97
|
-
- **Multimodal input via `context: { attachment: ... }`** — pass a file/IO/URL through `Step.run(input, context: { attachment: path })`; the adapter forwards it to `RubyLLM::Chat#ask(content, with: attachment)`. RubyLLM normalises wire format per provider (Anthropic url/base64, OpenAI `image_url`/`file`, Gemini `inline_data`). Multi-attachment supported natively (`with: [pdf1, pdf2]` or `with: { images: [...], pdfs: [...] }`). See [multimodal input guide](docs/guide/multimodal_input.md)
|
|
216
|
+
- **Multimodal input via `context: { attachment: ... }`** — pass a file/IO/URL through `Step.run(input, context: { attachment: path })`; the adapter forwards it to `RubyLLM::Chat#ask(content, with: attachment)`. RubyLLM normalises wire format per provider (Anthropic url/base64, OpenAI `image_url`/`file`, Gemini `inline_data`). Multi-attachment supported natively (`with: [pdf1, pdf2]` or `with: { images: [...], pdfs: [...] }`). See [multimodal input guide](docs/guide/multimodal_input.md) (rationale in the internal ADR-0022, not shipped).
|
|
98
217
|
- **`attachment_token_estimate(n)` class macro** — adopter-declared conservative estimate of attachment input tokens. Applied to BOTH runtime (`limit_checker`) and pre-flight (`estimate_cost`) — same source of truth, no estimate/runtime drift.
|
|
99
218
|
- **`on_unknown_attachment_size(:refuse | :warn)` class macro** — mirrors `on_unknown_pricing` opt-out semantics. Defaults to `:refuse`. Never settable as global default — same invariant as `max_cost` fail-closed.
|
|
100
219
|
|
|
@@ -274,7 +393,7 @@ end
|
|
|
274
393
|
|
|
275
394
|
### Documentation
|
|
276
395
|
|
|
277
|
-
- **Guide: [Production-mode cost measurement](docs/guide/optimizing_retry_policy.md#
|
|
396
|
+
- **Guide: [Production-mode cost measurement](docs/guide/optimizing_retry_policy.md#measure-effective-cost-before-shipping)** — API, metric interpretation, 2-tier scope note.
|
|
278
397
|
|
|
279
398
|
## 0.6.3 (2026-04-20)
|
|
280
399
|
|
|
@@ -283,7 +402,7 @@ end
|
|
|
283
402
|
- **`runs:` parameter on `compare_models` and `optimize_retry_policy`** — runs each candidate N times per eval and aggregates the mean score, mean cost per run, and mean latency. Reduces sampling variance in live mode where LLM outputs are non-deterministic (gpt-5 family enforces `temperature=1.0` server-side, so a single unlucky sample can misclassify a viable candidate as "failing"). Default `runs: 1` — backward compatible.
|
|
284
403
|
- **`RUNS=N` on `rake ruby_llm_contract:optimize`** — CLI flag for variance-aware optimization.
|
|
285
404
|
- **`Eval::AggregatedReport`** — duck-type `Report` exposing `score` (mean), `score_min`/`score_max` (spread), `total_cost` (mean per run), `pass_rate` (clean-pass count x/N), and `clean_passes`.
|
|
286
|
-
- **Guide: [Reducing variance with `runs:`](docs/guide/optimizing_retry_policy.md
|
|
405
|
+
- **Guide: [Reducing variance with `runs:`](docs/guide/optimizing_retry_policy.md)** — when to use it and why.
|
|
287
406
|
|
|
288
407
|
## 0.6.2 (2026-04-18)
|
|
289
408
|
|
|
@@ -305,9 +424,9 @@ end
|
|
|
305
424
|
|
|
306
425
|
### Features
|
|
307
426
|
|
|
308
|
-
- **Multi-provider operator tooling** — rake tasks support `PROVIDER=openai|anthropic|ollama
|
|
427
|
+
- **Multi-provider operator tooling** — rake tasks support `PROVIDER=openai|anthropic|ollama` and `CANDIDATES=model@effort,...`.
|
|
309
428
|
- **`rake ruby_llm_contract:recommend`** — wraps `Step.recommend` with CLI interface, prints best config, retry chain, DSL, rationale, and savings.
|
|
310
|
-
- **Ollama support** — `PROVIDER=ollama`
|
|
429
|
+
- **Ollama support** — `PROVIDER=ollama` (base URL via `RubyLLM.configure { |c| c.ollama_api_base = ... }`).
|
|
311
430
|
|
|
312
431
|
## 0.6.0 (2026-04-12)
|
|
313
432
|
|
data/README.md
CHANGED
|
@@ -23,7 +23,7 @@ end
|
|
|
23
23
|
RubyLLM::Contract.configure { }
|
|
24
24
|
```
|
|
25
25
|
|
|
26
|
-
Works with any `ruby_llm` provider (OpenAI, Anthropic, Gemini, etc). Requires `ruby_llm ~>
|
|
26
|
+
Works with any `ruby_llm` provider (OpenAI, Anthropic, Gemini, etc). Requires `ruby_llm ~> 2.0` and Ruby ≥ 3.2.
|
|
27
27
|
|
|
28
28
|
## Example
|
|
29
29
|
|
|
@@ -132,6 +132,8 @@ Everything below is optional — the example above is a complete step. Reach for
|
|
|
132
132
|
|
|
133
133
|
Also supports [multi-step pipelines](docs/guide/pipeline.md) with fail-fast and per-step models.
|
|
134
134
|
|
|
135
|
+
**Runnable companion repo:** [`ruby_llm-contract_demo`](https://github.com/justi/ruby_llm-contract_demo) — full LLM-as-judge lifecycle (v1 strict → v2 drift → judge calibration → v3/v4 iteration) on a returns-policy chatbot scenario, dual-language (EN default, `DEMO_LANG=pl` opt-in). Live OpenAI calls (`gpt-4.1-mini`, `temperature 0`), `LIVE=1` opt-in, ~$0.30 for the full lifecycle. See [llm_judge.md](docs/guide/llm_judge.md) for the methodology.
|
|
136
|
+
|
|
135
137
|
## Relation to `RubyLLM::Agent`
|
|
136
138
|
|
|
137
139
|
`Step::Base` and `RubyLLM::Agent` (since RubyLLM 1.12) are **siblings** targeting the same niche: reusable, class-based prompts. Both call into `RubyLLM::Chat` directly — Step does not wrap Agent. Step adds the contract layer: `validate` (business invariants), `retry_policy escalate(...)` (model escalation on validation failure), `max_cost` pre-flight refusal, regression-eval framework, pipeline composition. **[Full feature mapping →](docs/guide/relation_to_agent.md)**
|
|
@@ -162,10 +164,18 @@ Different layers, complementary. [`ruby_llm-tribunal`](https://github.com/Alqemi
|
|
|
162
164
|
| [Multimodal input (PDF / image / audio)](docs/guide/multimodal_input.md) | Route attachments through the contract; `attachment_token_estimate`, fail-closed cost, calibration table |
|
|
163
165
|
| [Schema DSL reference](docs/guide/output_schema.md) | Every constraint, nested objects, pattern table |
|
|
164
166
|
| [Prompt DSL reference](docs/guide/prompt_ast.md) | `system` / `rule` / `section` / `example` / `user` nodes |
|
|
167
|
+
| [Architecture overview](docs/architecture.md) | Class map: Pipeline, Step, Contract, Prompt, Eval |
|
|
165
168
|
|
|
166
169
|
## Status & versioning
|
|
167
170
|
|
|
168
|
-
|
|
171
|
+
Stable at **1.0.0**. Semver tracked; breaking changes flagged in [CHANGELOG](CHANGELOG.md). Pin `~> 1.0`.
|
|
172
|
+
|
|
173
|
+
What 1.0 promises not to break inside the 1.x line: the documented DSL
|
|
174
|
+
(`prompt`, `output_schema`, `validate`, `retry_policy`, `max_cost`, `define_eval`,
|
|
175
|
+
pipelines), the `Result` and `trace` shapes including `trace[:usage]`'s
|
|
176
|
+
`{ input_tokens:, output_tokens: }` keys, and the adapter interface. A new major
|
|
177
|
+
version of a runtime dependency may require a new major version here — that is
|
|
178
|
+
what 1.0.0 itself was.
|
|
169
179
|
|
|
170
180
|
## FAQ
|
|
171
181
|
|
data/docs/guide/llm_judge.md
CHANGED
|
@@ -120,7 +120,7 @@ report = AccuracyJudge.run_eval("calibration")
|
|
|
120
120
|
report.score # => 0.92 → judge agrees with humans 92% of the time
|
|
121
121
|
```
|
|
122
122
|
|
|
123
|
-
A reasonable bar is `score >= 0.85` —
|
|
123
|
+
A reasonable bar is `score >= 0.85` — a practical threshold synthesized from field practice. [Hamel Husain](https://hamel.dev/blog/posts/field-guide/) reports it took **three iterations** of the judge prompt to reach ">90% agreement" with humans in a Honeycomb case study; [Eugene Yan](https://eugeneyan.com/writing/evals/) notes RAG-grounded systems still hold a 5-10% factual inconsistency rate even after good prompt engineering. The specific 0.85 number is the author's summary — neither source quotes it directly. Below ~0.85, judge-human agreement drops into a band where the judge's confident percentages stop matching what a human reviewer would say, and gate decisions become unreliable. If your domain is high-stakes (medical, legal, financial), raise the bar to 0.90+. Iterate the judge's prompt, or escalate to a stronger model, until the score on real production data crosses your bar.
|
|
124
124
|
|
|
125
125
|
**What "iterate the judge's prompt" actually looks like.** The first judge prompt almost always over-flags. Two or three iterations are normal before the judge is ready to gate anything. The common pattern:
|
|
126
126
|
|
|
@@ -195,9 +195,9 @@ What to do operationally:
|
|
|
195
195
|
|
|
196
196
|
The drop is your signal to refine the judge's prompt, not to lower the gate.
|
|
197
197
|
|
|
198
|
-
## When to
|
|
198
|
+
## When to reach for Tribunal instead
|
|
199
199
|
|
|
200
|
-
|
|
200
|
+
[`ruby_llm-tribunal`](https://github.com/Alqemist-labs/ruby_llm-tribunal) ships an off-the-shelf catalog of common LLM-as-judge assertions (`assert_faithful`, `assert_hallucination`, `assert_refusal`, `assert_no_pii`, etc.) — a shortcut when your check matches one of those domain-general categories. The methodology in this guide still applies: calibrate the judge against your human-labeled production data **before** trusting Tribunal's `default_threshold = 0.8`, refine the prompt when it over-flags, watch for the anti-patterns above. See [Relation to Tribunal](relation_to_tribunal.md) for the full positioning — what each gem documents (and doesn't), a concrete decision tree on catalog-vs-custom-judge, and three working integration patterns.
|
|
201
201
|
|
|
202
202
|
## See also
|
|
203
203
|
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
> Read this when your contract needs to send a PDF, image, or audio file to the LLM — not just text.
|
|
4
4
|
|
|
5
|
-
`ruby_llm-contract` 0.10.0+ routes attachments through the contract layer, so `max_cost`, `validate`, `retry_policy escalate(...)`, and trace observability still apply. The gem does **not** ship its own multimodal API — it forwards `with:` to `RubyLLM::Chat#ask`, which RubyLLM
|
|
5
|
+
`ruby_llm-contract` 0.10.0+ routes attachments through the contract layer, so `max_cost`, `validate`, `retry_policy escalate(...)`, and trace observability still apply. The gem does **not** ship its own multimodal API — it forwards `with:` to `RubyLLM::Chat#ask`, which RubyLLM normalises per provider (Anthropic, OpenAI, Gemini).
|
|
6
6
|
|
|
7
7
|
## Minimal example
|
|
8
8
|
|
|
@@ -130,7 +130,7 @@ Pick a value at or above the provider's worst-case. The estimate is a **floor fo
|
|
|
130
130
|
|
|
131
131
|
## Multi-turn caveat
|
|
132
132
|
|
|
133
|
-
If your contract uses history (`add_history`), attachments from prior turns are **not** replayed in 0.
|
|
133
|
+
If your contract uses history (`add_history`), attachments from prior turns are **not** replayed (still true in 1.0.x). Single-turn multimodal works; follow-up questions on the same document require additional work that is deferred to a later release. The rationale is recorded in the project's internal ADR-0022, which is not shipped with the gem.
|
|
134
134
|
|
|
135
135
|
## Provider notes
|
|
136
136
|
|
|
@@ -149,7 +149,8 @@ RSpec.describe ExtractInvoiceData do
|
|
|
149
149
|
it "forwards attachment to chat.ask" do
|
|
150
150
|
expect_any_instance_of(RubyLLM::Chat).to receive(:ask)
|
|
151
151
|
.with(anything, with: "fixtures/invoice.pdf")
|
|
152
|
-
.and_return(double(content: '{"vendor":"X",...}',
|
|
152
|
+
.and_return(double(content: '{"vendor":"X",...}',
|
|
153
|
+
tokens: RubyLLM::Tokens.new(input: 200, output: 50)))
|
|
153
154
|
|
|
154
155
|
result = described_class.run("extract", context: { attachment: "fixtures/invoice.pdf" })
|
|
155
156
|
expect(result.status).to eq(:ok)
|
data/docs/guide/output_schema.md
CHANGED
|
@@ -2,12 +2,12 @@
|
|
|
2
2
|
|
|
3
3
|
> Read this as a reference for the schema DSL — every constraint, nested objects, arrays of objects, the full pattern table.
|
|
4
4
|
|
|
5
|
-
Declare the expected output structure using [
|
|
5
|
+
Declare the expected output structure using the [schematist](https://github.com/crmne/schematist) DSL. The schema serves **two purposes**:
|
|
6
6
|
|
|
7
7
|
1. **Output validation** — replaces type and shape checks (enums, ranges, required fields). One declaration instead of many.
|
|
8
8
|
2. **Provider-side request** — with the RubyLLM adapter, the schema is sent to the LLM provider via `chat.with_schema(...)`, asking the model to return JSON matching the shape. Cheaper models sometimes ignore the request, which is why client-side validation (point 1) still matters.
|
|
9
9
|
|
|
10
|
-
> **Same DSL `RubyLLM::Agent.schema` accepts.** Both use `
|
|
10
|
+
> **Same DSL `RubyLLM::Agent.schema` accepts.** Both use `Schematist::Schema.create(&block)` under the hood (schematist is what the `ruby_llm-schema` gem became; RubyLLM 2.x depends on it directly). `Step.output_schema` is eager-compiled at class load and drives client-side validation; `Agent.schema` accepts a `Proc` for dynamic per-call schemas. See [Relation to Agent](relation_to_agent.md) for the full comparison.
|
|
11
11
|
|
|
12
12
|
All examples below extend the `SummarizeArticle` step from the [README](../../README.md).
|
|
13
13
|
|
|
@@ -73,6 +73,34 @@ Tribunal grades **a fixed set of cases on every PR** to catch quality regression
|
|
|
73
73
|
|
|
74
74
|
**Both.** You ship contracts in prod (Contract) AND want stronger CI signal beyond schema regression — judge-quality grading on a frozen dataset, plus adversarial red-team probes. Use Contract's `Step` to make the call, run it in `define_eval` over your dataset, and grade each case with Tribunal helpers in your spec or via the dataset's `evaluator:` proc.
|
|
75
75
|
|
|
76
|
+
### Tribunal's catalog vs Contract's `llm_judge.md` — concrete decision tree
|
|
77
|
+
|
|
78
|
+
If you specifically need an **LLM-as-judge** (a second LLM grading the first one's output), the decision is:
|
|
79
|
+
|
|
80
|
+
- **Reach for Tribunal's catalog** when your check is one of the well-defined, domain-general categories Tribunal ships: *"is this faithful to the retrieved context?"*, *"is this a refusal?"*, *"does this contain PII?"*, *"is this jailbreak-resistant?"*, *"hallucinated?"*, *"toxic?"*, *"biased?"*. One line in a spec, default threshold, no judge code to write or maintain. The judge prompt is baked into the gem.
|
|
81
|
+
- **Build a custom judge per [`llm_judge.md`](llm_judge.md)** when:
|
|
82
|
+
- **Your criterion is domain-specific** — *"does this medical advice match our internal safety policy?"*, *"is this reply in our brand voice?"*, *"does this summary preserve the legal disclaimer verbatim?"*. No off-the-shelf judge knows your policy; you write the prompt.
|
|
83
|
+
- **You need the verdict inside a `define_eval` regression gate** (the `evaluator:` lambda pattern) — Tribunal's surface is spec-time assertions, not eval-framework evaluators.
|
|
84
|
+
- **You need a per-claim breakdown** (sentence-level *"this claim → unsupported, that claim → contradicted"* output) for PR debugging — Tribunal returns one score per assertion.
|
|
85
|
+
- **You need to iterate the judge prompt** because it over-flags on your data — Tribunal's prompts are fixed per assertion.
|
|
86
|
+
- **Use both** for the same project even when your check is in Tribunal's catalog: Tribunal's `assert_faithful` for spec-time grade on individual responses, plus a calibrated custom judge wired as `evaluator:` in a regression `define_eval` over a frozen dataset for CI merge-gating. They cover different lifecycle stages.
|
|
87
|
+
|
|
88
|
+
Either way, the **methodology** in [`llm_judge.md`](llm_judge.md) — calibrate the judge against human-labeled production samples before trusting any score, watch for the six anti-patterns, refine the prompt when it over-flags — applies equally to Tribunal's built-ins, Tribunal's custom registered judges, and Contract `Step::Base` judges. Tribunal's `default_threshold = 0.8` is a starting point, not a calibrated bar for your data.
|
|
89
|
+
|
|
90
|
+
## What Tribunal documents — and what it doesn't
|
|
91
|
+
|
|
92
|
+
Tribunal's README ships an **implementation catalog** (`assert_faithful`, `assert_hallucination`, `assert_refusal`, `assert_no_pii`, `assert_no_toxicity`, `assert_no_bias`, `assert_jailbreak_resistant`, etc., plus a `register_judge` API for custom ones). What it currently leaves to the adopter:
|
|
93
|
+
|
|
94
|
+
| Tribunal ships | Tribunal's README doesn't document (Contract's [`llm_judge.md`](llm_judge.md) does) |
|
|
95
|
+
|---|---|
|
|
96
|
+
| `default_threshold = 0.8` (fixed) | How to **calibrate** the threshold against your human-labeled production data |
|
|
97
|
+
| Judge prompt baked in per assertion | How to **iterate the judge prompt** when it over-flags stylistic courtesy as drift |
|
|
98
|
+
| Single score per assertion | **Per-claim breakdown** schema for sentence-level PR debugging |
|
|
99
|
+
| `assert_faithful` in a spec | **Judge as `evaluator:` lambda** in a Contract `define_eval` regression gate |
|
|
100
|
+
| Custom Judge mechanism (`register_judge`) | **Anti-patterns** (stubbing the verdict, calibrating on synthetic data, calibrating once and shipping) |
|
|
101
|
+
|
|
102
|
+
This is a **complementary gap**, not a competition. Tribunal owns the implementation catalog; Contract's `llm_judge.md` owns the methodology. A typical production setup uses both layers: pick (or build) the implementation, then calibrate it against your humans **before** trusting any score.
|
|
103
|
+
|
|
76
104
|
## Integration patterns
|
|
77
105
|
|
|
78
106
|
These work today without any code changes in either gem — both use plain Ruby blocks/procs as extension points.
|
data/examples/README.md
CHANGED
|
@@ -14,7 +14,7 @@ Pedagogical order: hook → activation → evolution → composition → quality
|
|
|
14
14
|
| 05 | `05_eval_dataset.rb` | **"How do I stop silent prompt regressions?"** — define_eval with real cases, baseline vs regressed adapter, regression detection signal, inline eval_case. |
|
|
15
15
|
| 06 | `06_retry_variants.rb` | **"What retry shapes exist beyond cross-model?"** — `attempts: 3` (variance absorption), `reasoning_effort` escalation (low→medium→high), cross-provider fallback (Ollama → Anthropic → OpenAI). |
|
|
16
16
|
|
|
17
|
-
|
|
17
|
+
Four of the seven carry an "Expected output" section in the file header (01, 03, 05, 06), so you can read what they print without running them. The other three differ: `00_basics.rb` prints nothing and annotates each result inline with `# =>`; `02_real_llm_minimal.rb` needs a live provider key, so its output is not deterministic; `04_summarize_and_translate.rb` prints but has no header block.
|
|
18
18
|
|
|
19
19
|
## Running
|
|
20
20
|
|
|
@@ -25,7 +25,6 @@ module RubyLLM
|
|
|
25
25
|
build_response(response)
|
|
26
26
|
end
|
|
27
27
|
|
|
28
|
-
# Maps option keys to the RubyLLM chat method and argument form.
|
|
29
28
|
CHAT_OPTION_METHODS = {
|
|
30
29
|
temperature: :with_temperature,
|
|
31
30
|
schema: :with_schema
|
|
@@ -70,12 +69,10 @@ module RubyLLM
|
|
|
70
69
|
thinking_config = resolve_thinking_config(options)
|
|
71
70
|
chat.with_thinking(**thinking_config) if thinking_config
|
|
72
71
|
|
|
73
|
-
#
|
|
74
|
-
#
|
|
75
|
-
#
|
|
76
|
-
|
|
77
|
-
params[:max_tokens] = options[:max_tokens] if options[:max_tokens]
|
|
78
|
-
chat.with_params(**params) if params.any?
|
|
72
|
+
# RubyLLM 2.0 removed `with_params`; `with_max_output_tokens` is the
|
|
73
|
+
# replacement for this one passthrough. `reasoning_effort` is not
|
|
74
|
+
# forwarded here — it goes through `with_thinking` above.
|
|
75
|
+
chat.with_max_output_tokens(options[:max_tokens]) if options[:max_tokens]
|
|
79
76
|
end
|
|
80
77
|
|
|
81
78
|
# Returns merged `{ effort:, budget: }` or nil. `options[:reasoning_effort]`
|
|
@@ -91,11 +88,18 @@ module RubyLLM
|
|
|
91
88
|
content = response.content
|
|
92
89
|
content = content.to_s unless content.is_a?(Hash) || content.is_a?(Array)
|
|
93
90
|
|
|
91
|
+
# This is the ONLY place upstream token counts are read. RubyLLM 2.0
|
|
92
|
+
# replaced `Message#input_tokens`/`#output_tokens` with a `Tokens` value
|
|
93
|
+
# object. The `{ input_tokens:, output_tokens: }` shape below is this
|
|
94
|
+
# gem's own public contract (Step::Trace#usage, cost calculation,
|
|
95
|
+
# documented in the README) and deliberately does NOT change.
|
|
96
|
+
tokens = response.tokens
|
|
97
|
+
|
|
94
98
|
Response.new(
|
|
95
99
|
content: content,
|
|
96
100
|
usage: {
|
|
97
|
-
input_tokens:
|
|
98
|
-
output_tokens:
|
|
101
|
+
input_tokens: tokens&.input || 0,
|
|
102
|
+
output_tokens: tokens&.output || 0
|
|
99
103
|
}
|
|
100
104
|
)
|
|
101
105
|
end
|
|
@@ -38,7 +38,6 @@ module RubyLLM
|
|
|
38
38
|
end
|
|
39
39
|
private_class_method :parse_json_text
|
|
40
40
|
|
|
41
|
-
# Fallback: attempt to extract the first JSON object or array from prose
|
|
42
41
|
def self.parse_json_with_extraction(text, raw_output)
|
|
43
42
|
extracted = extract_json(text)
|
|
44
43
|
unless extracted
|
|
@@ -72,8 +71,6 @@ module RubyLLM
|
|
|
72
71
|
match ? match[1] : text
|
|
73
72
|
end
|
|
74
73
|
|
|
75
|
-
# Extract the first JSON object or array from text that may contain prose.
|
|
76
|
-
# Uses bracket-matching to find the outermost balanced { } or [ ] block.
|
|
77
74
|
JSON_START_PATTERN = /[{\[]/
|
|
78
75
|
|
|
79
76
|
def self.extract_json(text)
|
|
@@ -84,11 +84,36 @@ module RubyLLM
|
|
|
84
84
|
end
|
|
85
85
|
|
|
86
86
|
def self.compute_cost(model_info, usage)
|
|
87
|
-
|
|
88
|
-
|
|
87
|
+
prices = prices_for(model_info)
|
|
88
|
+
return nil unless prices
|
|
89
|
+
|
|
90
|
+
input_cost = token_cost(usage[:input_tokens], prices[:input])
|
|
91
|
+
output_cost = token_cost(usage[:output_tokens], prices[:output])
|
|
89
92
|
(input_cost + output_cost).round(6)
|
|
90
93
|
end
|
|
91
94
|
|
|
95
|
+
# Two shapes reach here. `RegisteredModel` (our own struct, from
|
|
96
|
+
# `register_model`) exposes flat `*_price_per_million` readers. RubyLLM 2.0
|
|
97
|
+
# moved provider pricing into a nested value object and dropped those flat
|
|
98
|
+
# readers, so it has to be walked: pricing -> text_tokens -> standard.
|
|
99
|
+
#
|
|
100
|
+
# nil means "unknown pricing", which `max_cost` fails closed on. The guards
|
|
101
|
+
# keep the two known shapes explicit, but `calculate` rescues StandardError,
|
|
102
|
+
# so a future shape this walk cannot read still degrades to nil rather than
|
|
103
|
+
# surfacing - every budgeted call would be refused with no error.
|
|
104
|
+
def self.prices_for(model_info)
|
|
105
|
+
if model_info.respond_to?(:input_price_per_million)
|
|
106
|
+
return { input: model_info.input_price_per_million,
|
|
107
|
+
output: model_info.output_price_per_million }
|
|
108
|
+
end
|
|
109
|
+
|
|
110
|
+
tier = model_info.pricing&.text_tokens&.standard if model_info.respond_to?(:pricing)
|
|
111
|
+
return unless tier.respond_to?(:input_per_million)
|
|
112
|
+
|
|
113
|
+
{ input: tier.input_per_million, output: tier.output_per_million }
|
|
114
|
+
end
|
|
115
|
+
private_class_method :prices_for
|
|
116
|
+
|
|
92
117
|
# Provider pricing is denominated per 1M tokens; divide here to get
|
|
93
118
|
# the dollar cost for the actual usage count. Named constant for
|
|
94
119
|
# consistency with how RubyLLM and provider docs express prices.
|
|
@@ -79,8 +79,10 @@ module RubyLLM
|
|
|
79
79
|
private
|
|
80
80
|
|
|
81
81
|
def compute_score(cases)
|
|
82
|
-
#
|
|
83
|
-
|
|
82
|
+
# Asymmetric on purpose: the baseline side arrives already filtered by
|
|
83
|
+
# ReportStats#evaluated_results, the current side does not, so this
|
|
84
|
+
# string check is the only skip filter on one of the two sides.
|
|
85
|
+
evaluated = cases.reject { |c| c[:details]&.start_with?(CaseResult::SKIPPED_DETAILS_PREFIX) }
|
|
84
86
|
return 0.0 if evaluated.empty?
|
|
85
87
|
|
|
86
88
|
evaluated.sum { |c| c[:score] } / evaluated.length
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module RubyLLM
|
|
4
|
+
module Contract
|
|
5
|
+
module Eval
|
|
6
|
+
# The `model (effort: x)` notation, rendered into optimizer tables and parsed
|
|
7
|
+
# back to rebuild candidate configs. Render and parse live together because a
|
|
8
|
+
# format change on one side silently mis-parses on the other: the pattern is
|
|
9
|
+
# built from the same literal the renderer emits.
|
|
10
|
+
module CandidateLabel
|
|
11
|
+
OPEN = " (effort: "
|
|
12
|
+
CLOSE = ")"
|
|
13
|
+
PATTERN = /\s*#{Regexp.escape(OPEN.lstrip)}(\w+)#{Regexp.escape(CLOSE)}/
|
|
14
|
+
|
|
15
|
+
def self.render(config)
|
|
16
|
+
effort = config[:reasoning_effort]
|
|
17
|
+
return config[:model] unless effort
|
|
18
|
+
|
|
19
|
+
"#{config[:model]}#{OPEN}#{effort}#{CLOSE}"
|
|
20
|
+
end
|
|
21
|
+
|
|
22
|
+
def self.parse(label)
|
|
23
|
+
match = PATTERN.match(label)
|
|
24
|
+
return { model: label } unless match
|
|
25
|
+
|
|
26
|
+
{ model: match.pre_match.strip, reasoning_effort: match[1] }
|
|
27
|
+
end
|
|
28
|
+
end
|
|
29
|
+
end
|
|
30
|
+
end
|
|
31
|
+
end
|
|
@@ -30,7 +30,7 @@ module RubyLLM
|
|
|
30
30
|
private
|
|
31
31
|
|
|
32
32
|
def missing_adapter?(error)
|
|
33
|
-
error.message.include?(
|
|
33
|
+
error.message.include?(Step::Base::NO_ADAPTER_MESSAGE)
|
|
34
34
|
end
|
|
35
35
|
|
|
36
36
|
def skipped_result(test_case, reason)
|
|
@@ -43,7 +43,7 @@ module RubyLLM
|
|
|
43
43
|
score: 0.0,
|
|
44
44
|
passed: false,
|
|
45
45
|
label: "SKIP",
|
|
46
|
-
details:
|
|
46
|
+
details: CaseResult.skipped_details(reason)
|
|
47
47
|
)
|
|
48
48
|
end
|
|
49
49
|
end
|
|
@@ -4,6 +4,14 @@ module RubyLLM
|
|
|
4
4
|
module Contract
|
|
5
5
|
module Eval
|
|
6
6
|
class CaseResult
|
|
7
|
+
# BaselineDiff#compute_score keys off this prefix to keep skipped cases
|
|
8
|
+
# out of the score denominator. Both ends must read it from here.
|
|
9
|
+
SKIPPED_DETAILS_PREFIX = "skipped:"
|
|
10
|
+
|
|
11
|
+
def self.skipped_details(reason)
|
|
12
|
+
"#{SKIPPED_DETAILS_PREFIX} #{reason}"
|
|
13
|
+
end
|
|
14
|
+
|
|
7
15
|
attr_reader :name, :input, :output, :expected, :step_status,
|
|
8
16
|
:score, :details, :duration_ms, :cost, :attempts
|
|
9
17
|
|
|
@@ -6,9 +6,10 @@ module RubyLLM
|
|
|
6
6
|
class ModelComparison
|
|
7
7
|
attr_reader :eval_name, :reports, :configs, :fallback
|
|
8
8
|
|
|
9
|
+
# Kept as the name six call sites already use; the format itself lives in
|
|
10
|
+
# CandidateLabel, which owns both directions.
|
|
9
11
|
def self.candidate_label(config)
|
|
10
|
-
|
|
11
|
-
effort ? "#{config[:model]} (effort: #{effort})" : config[:model]
|
|
12
|
+
CandidateLabel.render(config)
|
|
12
13
|
end
|
|
13
14
|
|
|
14
15
|
def initialize(eval_name:, reports:, configs: nil, fallback: nil)
|