ruby_llm-contract 0.10.5 → 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +124 -5
  3. data/README.md +12 -2
  4. data/docs/guide/llm_judge.md +3 -3
  5. data/docs/guide/multimodal_input.md +4 -3
  6. data/docs/guide/output_schema.md +2 -2
  7. data/docs/guide/relation_to_tribunal.md +28 -0
  8. data/examples/README.md +1 -1
  9. data/lib/ruby_llm/contract/adapters/ruby_llm.rb +13 -9
  10. data/lib/ruby_llm/contract/concerns/context_helpers.rb +0 -2
  11. data/lib/ruby_llm/contract/concerns/deep_symbolize.rb +0 -1
  12. data/lib/ruby_llm/contract/contract/parser.rb +0 -3
  13. data/lib/ruby_llm/contract/contract/schema_validator/bound_rule.rb +0 -1
  14. data/lib/ruby_llm/contract/contract/schema_validator/enum_rule.rb +0 -1
  15. data/lib/ruby_llm/contract/contract/schema_validator/node.rb +0 -4
  16. data/lib/ruby_llm/contract/contract/schema_validator/scalar_rules.rb +0 -1
  17. data/lib/ruby_llm/contract/contract/schema_validator/type_rule.rb +0 -1
  18. data/lib/ruby_llm/contract/cost_calculator.rb +27 -2
  19. data/lib/ruby_llm/contract/eval/baseline_diff.rb +4 -2
  20. data/lib/ruby_llm/contract/eval/candidate_label.rb +31 -0
  21. data/lib/ruby_llm/contract/eval/case_executor.rb +2 -2
  22. data/lib/ruby_llm/contract/eval/case_result.rb +8 -0
  23. data/lib/ruby_llm/contract/eval/contract_detail_builder.rb +0 -2
  24. data/lib/ruby_llm/contract/eval/model_comparison.rb +3 -2
  25. data/lib/ruby_llm/contract/eval/pipeline_result_adapter.rb +1 -2
  26. data/lib/ruby_llm/contract/eval/prompt_diff_comparator.rb +0 -1
  27. data/lib/ruby_llm/contract/eval/prompt_diff_presenter.rb +0 -1
  28. data/lib/ruby_llm/contract/eval/prompt_diff_serializer.rb +0 -1
  29. data/lib/ruby_llm/contract/eval/report_presenter.rb +0 -1
  30. data/lib/ruby_llm/contract/eval/report_stats.rb +0 -1
  31. data/lib/ruby_llm/contract/eval/report_storage.rb +0 -1
  32. data/lib/ruby_llm/contract/eval/retry_optimizer.rb +10 -20
  33. data/lib/ruby_llm/contract/eval/trait_evaluator.rb +0 -2
  34. data/lib/ruby_llm/contract/eval.rb +1 -0
  35. data/lib/ruby_llm/contract/pipeline/runner.rb +0 -1
  36. data/lib/ruby_llm/contract/rake_task/suite_gate.rb +0 -12
  37. data/lib/ruby_llm/contract/rake_task.rb +1 -2
  38. data/lib/ruby_llm/contract/step/base.rb +12 -3
  39. data/lib/ruby_llm/contract/step/dsl.rb +2 -4
  40. data/lib/ruby_llm/contract/step/limit_checker.rb +0 -2
  41. data/lib/ruby_llm/contract/step/retry_executor.rb +0 -2
  42. data/lib/ruby_llm/contract/token_estimator.rb +0 -2
  43. data/lib/ruby_llm/contract/version.rb +1 -1
  44. data/lib/ruby_llm/contract.rb +0 -4
  45. data/ruby_llm-contract.gemspec +6 -5
  46. metadata +7 -11
  47. data/.rubocop.yml +0 -58
  48. data/Gemfile +0 -13
  49. data/Gemfile.lock +0 -278
  50. data/Rakefile +0 -8
checksums.yaml CHANGED
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  SHA256:
3
- metadata.gz: 233b114efc88c168573b434e9adc17163401e747920a4fc6b587e7a834721723
4
- data.tar.gz: fd2d57bd72cd72a7b4a405fda8bfcb0865d737a16a78db11e69d25026c89d1a2
3
+ metadata.gz: d7d18fbcf79a634c5f577d8965bc82ed10c12a8593b244ad3f07c28f21033035
4
+ data.tar.gz: '009718225fb17d5b34ea3fe3bf290b812fb17321faa1278d027ae8bdc66e9c2c'
5
5
  SHA512:
6
- metadata.gz: ff1564d786bc8ab8b391d3dfe0e7e9be406dd4bcaca56fc891fa3ead2a3b8c76cfb6f26c3ad81bc2ffd798e3b9a63a90ab1c4a847e1ad2be1efc1170c6abee83
7
- data.tar.gz: 33a4a6aa6798f55d6cfe9c4a1684ee3080e1ea42043d7f53ab54b40c5cc927a9dd607a266a3130206b43822ae7c06739856d29517237432d6cdcfda2f1b08ff3
6
+ metadata.gz: 2db5376a4aefe02acd3ce2a2debc645d2c2b14d185877da6cbccd9810b04aea8f1a3c830d48d693a66ae8268774ff5041ee81d3c3eba2e33cdc22e5d0ec15d0e
7
+ data.tar.gz: f560b145abc0b5114240edec2cc01fe563b9d51f48ef90c80a72e9b6b121e3246b7c9ac3196027471bd5450b3b3225f53f566d2d15c50976b8106fda889a458a
data/CHANGELOG.md CHANGED
@@ -1,5 +1,124 @@
1
1
  # Changelog
2
2
 
3
+ ## 1.0.0 (2026-09-27)
4
+
5
+ **Requires `ruby_llm ~> 2.0`.** This is the only breaking change: it is a
6
+ dependency requirement, not a change to this gem's API. Your call sites,
7
+ `output_schema` blocks, `trace[:usage]` keys and adapter interface are all
8
+ untouched — if you are on ruby_llm 2.x, upgrading should need no code edits.
9
+
10
+ 1.0 also states what is now stable: the documented DSL, the `Result`/`trace`
11
+ shapes including `trace[:usage]`'s `{ input_tokens:, output_tokens: }` keys, and
12
+ the adapter interface. A future major version of a runtime dependency may again
13
+ require a major version here — that is what this release is.
14
+
15
+ ### Added
16
+
17
+ - **CI, for the first time.** Matrix over Ruby 3.2 and 3.4 against both
18
+ ruby_llm 2.0.0 (pinned, so a regression at the major boundary stays
19
+ distinguishable) and the newest 2.x the gemspec allows. The job also runs every
20
+ offline example, builds the gem and fails on any `gem build` warning, and
21
+ installs the built gem to prove it `require`s. The seven live-key specs are
22
+ bounded explicitly: CI asserts the pending count is exactly 7 and that all
23
+ seven come from `spec/integration/cost_of_quality_real_spec.rb`, so a failing
24
+ spec cannot be quietly downgraded to pending.
25
+ - **`.rubocop_todo.yml`** so lint can be enforced in CI at zero offences while the
26
+ pre-existing backlog stays visible and shrinkable, rather than blocking the gate.
27
+ - **A boundary spec for the RubyLLM 2.x seam**
28
+ (`spec/ruby_llm/contract/adapters/ruby_llm_2x_boundary_spec.rb`). Every
29
+ upstream-version-specific detail is supposed to live in `Adapters::RubyLLM` and
30
+ `CostCalculator`; these tests pin that, including that `max_cost` still fails
31
+ closed if provider pricing ever becomes unreachable again.
32
+
33
+ ### Changed
34
+
35
+ - **`ruby_llm` requirement raised to `~> 2.0`** (from `~> 1.12`).
36
+ - **`ruby_llm-schema` replaced by `schematist ~> 1.1`.** `ruby_llm-schema` 1.0.0 is
37
+ 31 lines that emit a deprecation warning and alias `RubyLLM::Schema =
38
+ Schematist::Schema`; ruby_llm 2.x depends on schematist directly. The
39
+ `output_schema do ... end` DSL is unchanged — same object behind a different
40
+ constant.
41
+ - **Adapter reads token counts from `Message#tokens`.** RubyLLM 2.0 replaced
42
+ `Message#input_tokens`/`#output_tokens` with a `Tokens` value object. This is the
43
+ only place in the gem that touches upstream token counts, and the normalised
44
+ `{ input_tokens:, output_tokens: }` hash it publishes is deliberately unchanged.
45
+ - **`max_tokens` now forwards via `with_max_output_tokens`**; `Chat#with_params` no
46
+ longer exists in 2.0.
47
+
48
+ ### Fixed
49
+
50
+ - **`estimate_eval_cost` raised `TypeError` instead of flooring at $0.00.** With
51
+ ruby_llm 2.x, `CostCalculator#calculate` returns nil for two distinct misses -
52
+ model absent from the registry, and model present but with unreadable pricing -
53
+ and the summation guarded only the first. Measured against the real 2.0.0
54
+ registry, 669 of 1672 models take the second path, so any eval estimate touching
55
+ one of them crashed. `getting_started.md` documents this method as a floor, not
56
+ a fail-closed, and it behaves that way again.
57
+ - **The optimizer's candidate table could mangle a model name containing
58
+ brackets.** The short label was built by cutting the rendered label apart, so
59
+ `"custom (beta) (effort: high)"` came out as `"custom (beta@high)"` - the
60
+ model's own closing bracket was consumed. The notation now has one owner
61
+ (`Eval::CandidateLabel`, which both renders and parses), and the short form is
62
+ composed from parsed parts instead of string surgery.
63
+ - **A reworded skipped-case reason could have reported a phantom regression.**
64
+ `BaselineDiff` excludes skipped cases from the score denominator by matching a
65
+ string prefix that `CaseExecutor` writes, and the two sides are filtered
66
+ differently - the baseline side structurally, the current side only by that
67
+ string. Both ends now read one constant, and a baseline file written by 0.10.6
68
+ is covered by its own spec (the fixture was generated by installing 0.10.6, not
69
+ written by hand).
70
+ - **Cost calculation against ruby_llm 2.x.** 2.0 dropped the flat
71
+ `input_price_per_million` / `output_price_per_million` readers on `Model` for a
72
+ nested `pricing -> text_tokens -> standard` walk. Because `CostCalculator`
73
+ rescued `StandardError` into a nil cost, every priced model silently became
74
+ "unknown pricing" — and since `max_cost` fails closed on unknown pricing, every
75
+ budgeted call would have been refused. Pricing extraction is now explicit for
76
+ both shapes (upstream models and locally `register_model`-ed ones), and an
77
+ unrecognised shape returns nil rather than a swallowed exception.
78
+ - **A shipped doc example taught the wrong test double.**
79
+ `docs/guide/multimodal_input.md` showed `double(content:, input_tokens:,
80
+ output_tokens:)`, which a verified double rejects against ruby_llm 2.0.
81
+ - **`FINDING 7` in the audit specs asserted a constraint on `ruby_llm-schema`**, a
82
+ dependency this release removes. Generalised to "no runtime dependency is
83
+ declared without a version constraint", so the check cannot be retired by
84
+ renaming the gem it happened to be about.
85
+ - Stale version and dependency claims in `README.md`, `docs/guide/output_schema.md`
86
+ and `docs/guide/multimodal_input.md`.
87
+
88
+ ### Housekeeping (was staged as 0.11.0)
89
+
90
+ Result of a 20-pass systematic audit of the whole repository. No API changes and no behaviour change inside `Step`/`Pipeline`/`Eval`; the user-visible differences are in what the gem **packages**, what the RubyGems page **links**, and what the docs **claim**.
91
+
92
+ ### Changed
93
+
94
+ - **The published gem no longer ships development-only files.** Measured by building the gem and unpacking it: `.rubocop.yml`, `Gemfile`, `Gemfile.lock` and `Rakefile` were being packaged, contradicting the gemspec's own stated policy ("dev configs excluded so the published gem contains only what adopters actually need at runtime"). The shipped `Rakefile` was unusable anyway — it `require`s rspec, which is not a runtime dependency, and defines a task over `spec/`, which the same build excludes. Package contents: **137 → 133 files**; `lib/` (105) and `docs/` (16) unchanged.
95
+ - **`Gemfile.lock` is no longer tracked.** It was in the index despite being matched by `.gitignore:10`, and was repacked on every release. A library should not pin its adopters' dependency versions.
96
+ - **The RubyGems page will now show a "Source Code" link.** `metadata["homepage_uri"]` duplicated `spec.homepage` and, being listed first, suppressed `source_code_uri` in the UI — `gem build` warned about this on every build. Dropped the duplicate; `source_code_uri` stays, because `bundle info` and the RubyGems API read it as a distinct field.
97
+
98
+ ### Fixed
99
+
100
+ - **`.rubocop.yml` declared `AllCops` twice.** Psych applies last-key-wins, so `TargetRubyVersion: 3.2`, `NewCops: enable` and `SuggestExtensions: false` were silently discarded, and the second block's `Exclude` **replaced** RuboCop's defaults instead of adding to them. Effect: the linter scanned 353 files / 2423 offences, 147 of those files untracked under `tmp/` (they carried 2155 of the offences). Merging the blocks alone does not fix it — `inherit_mode: merge` is what restores the defaults. Now 206 files / 300 offences, all pre-existing formatting on real code. (Both counts measured with rubocop 1.88.0; `Gemfile.lock` is no longer tracked, so pin the tool when reproducing.)
101
+ - **`docs/guide/multimodal_input.md` linked to a file that could never resolve** — `../decisions/ADR-0022-*.md` points at `docs/decisions/` (no such directory; the ADR lives in `doc/decisions/`, singular) and that path is gitignored. Since `docs/guide/*` is packaged, this reproduced exactly the "links 404'd locally" defect that 0.10.2 set out to fix. Same dangling reference removed from the 0.9.0 CHANGELOG entry.
102
+ - **Two documented rake-task environment variables never existed.** `REASONING_EFFORT=` and `OLLAMA_API_BASE=` were listed in the 0.5.0 CHANGELOG entry, but `git log --all -S<name> -- lib/` is empty for both — they were never read at any point in the project's history, and ruby_llm's own configuration only seeds `RUBYLLM_DEBUG` / `RUBYLLM_STREAM_DEBUG` from the environment. Setting either was a silent no-op. Reasoning effort is selected with `CANDIDATES=model@effort`; Ollama's base URL goes through `RubyLLM.configure`.
103
+ - **README stated the wrong current version** (0.10.4 while `version.rb` was 0.10.6) — the one sentence that has to change every release.
104
+ - **`examples/README.md` claimed every example carries an "Expected output" section**; four of seven do, for three different reasons, so the sentence was the error rather than the files.
105
+ - **`docs/architecture.md` was reachable from nowhere** — the only one of 16 tracked docs with no inbound link. Added to the README index.
106
+
107
+ ### Removed
108
+
109
+ - Dead code with a full evidence battery behind each cut: `SchemaValidator::Node#numeric?` (zero readers; the one place needing the check inlines `value.is_a?(Numeric)`), an abandoned twin test helper whose sibling has 14 call sites, three unused `let`/helper definitions, four useless assignments, seven redundant `require`s in specs, and a `SimpleCov.start` that was a measured no-op because `require "simplecov"` already autoloads `.simplecov`.
110
+ - 31 WHAT comment blocks (51 lines) in `lib/` plus 20 comment lines in `spec/`, all restating the line or class below them - including all six `Extracted from ...` comments, five of them the "to reduce class length" boilerplate (refactor history git already records) and ten one-line class docstrings paraphrasing their own class name. WHY comments — invariants, external-behaviour notes, anti-duplication markers — were deliberately kept; so were the three evaluator banners that name a comparison semantic the class name cannot convey.
111
+ - Private-project residue: a spec block still labelled "reddit promo planner shape" although that project was removed as private in #25, and four dangling `(Batch N / TODO)` references to a `TODO.md` deleted in 0.10.1.
112
+
113
+ ## 0.10.6 (2026-06-11)
114
+
115
+ Docs accuracy patch: correct the positioning of `llm_judge.md` against `ruby_llm-tribunal` after a deeper audit of Tribunal's documented scope. The previous wording ("you are reinventing what `ruby_llm-tribunal` ships as a built-in catalog") read as if Tribunal made `llm_judge.md` redundant — incorrect. Tribunal's README ships an off-the-shelf implementation catalog (`assert_faithful`, `assert_hallucination`, `assert_refusal`, `assert_no_pii`, etc.) but does **not** document calibration workflow, per-claim breakdown, judge-prompt iteration, or judge-as-`evaluator:` integration — exactly the methodology `llm_judge.md` covers. The two are complementary layers, not alternatives. No code behaviour change.
116
+
117
+ ### Fixed
118
+
119
+ - **`docs/guide/relation_to_tribunal.md`** — added the **"Tribunal's catalog vs Contract's `llm_judge.md` — concrete decision tree"** sub-section under "When to use which", giving adopters a sharp three-way fork: reach for Tribunal's catalog when the check is domain-general (faithfulness vs context, hallucination, refusal, PII, jailbreak, toxicity, bias); build a custom judge per `llm_judge.md` when the criterion is domain-specific, when the judge needs to live inside a `define_eval` regression gate as the `evaluator:` lambda, when per-claim sentence-level debug output is required, or when the judge prompt itself needs to be iterated against your data; use both for the same project at different lifecycle stages (Tribunal at spec-time, calibrated custom judge at CI merge gate). Added the **"What Tribunal documents — and what it doesn't"** sub-section with a five-row comparison table making explicit which methodology gaps `llm_judge.md` covers that Tribunal's README leaves to the adopter (calibration against humans, prompt iteration on over-flag, per-claim breakdown, evaluator-lambda integration, the six anti-patterns).
120
+ - **`docs/guide/llm_judge.md`** — rewrote the closing "When to escalate to Tribunal's catalog" section as **"When to reach for Tribunal instead"**: shorter, accurate (Tribunal is a complementary catalog, not a replacement), points to `relation_to_tribunal.md` for the full decision tree and integration patterns. The previous wording implied that building any of the four standard judges (faithful / hallucination / refusal / PII) was "reinvention" — true for the **implementation** (Tribunal ships them), false for the **methodology** (Tribunal's README doesn't document calibration, anti-patterns, or per-claim breakdown). The methodology applies equally to Tribunal's built-ins, Tribunal's custom registered judges, and Contract `Step::Base` judges.
121
+
3
122
  ## 0.10.5 (2026-06-11)
4
123
 
5
124
  Docs release: new `llm_judge.md` guide + comprehensive clarity audit across all 16 shipping documentation files. No code behaviour change.
@@ -94,7 +213,7 @@ validate("score in range 0-100") { |o| o[:score].between?(0, 100) }
94
213
 
95
214
  ### Added
96
215
 
97
- - **Multimodal input via `context: { attachment: ... }`** — pass a file/IO/URL through `Step.run(input, context: { attachment: path })`; the adapter forwards it to `RubyLLM::Chat#ask(content, with: attachment)`. RubyLLM normalises wire format per provider (Anthropic url/base64, OpenAI `image_url`/`file`, Gemini `inline_data`). Multi-attachment supported natively (`with: [pdf1, pdf2]` or `with: { images: [...], pdfs: [...] }`). See [multimodal input guide](docs/guide/multimodal_input.md) and [ADR-0022](doc/decisions/ADR-0022-v09-multimodal-input.md).
216
+ - **Multimodal input via `context: { attachment: ... }`** — pass a file/IO/URL through `Step.run(input, context: { attachment: path })`; the adapter forwards it to `RubyLLM::Chat#ask(content, with: attachment)`. RubyLLM normalises wire format per provider (Anthropic url/base64, OpenAI `image_url`/`file`, Gemini `inline_data`). Multi-attachment supported natively (`with: [pdf1, pdf2]` or `with: { images: [...], pdfs: [...] }`). See [multimodal input guide](docs/guide/multimodal_input.md) (rationale in the internal ADR-0022, not shipped).
98
217
  - **`attachment_token_estimate(n)` class macro** — adopter-declared conservative estimate of attachment input tokens. Applied to BOTH runtime (`limit_checker`) and pre-flight (`estimate_cost`) — same source of truth, no estimate/runtime drift.
99
218
  - **`on_unknown_attachment_size(:refuse | :warn)` class macro** — mirrors `on_unknown_pricing` opt-out semantics. Defaults to `:refuse`. Never settable as global default — same invariant as `max_cost` fail-closed.
100
219
 
@@ -274,7 +393,7 @@ end
274
393
 
275
394
  ### Documentation
276
395
 
277
- - **Guide: [Production-mode cost measurement](docs/guide/optimizing_retry_policy.md#production-mode-cost-measurement)** — API, metric interpretation, 2-tier scope note.
396
+ - **Guide: [Production-mode cost measurement](docs/guide/optimizing_retry_policy.md#measure-effective-cost-before-shipping)** — API, metric interpretation, 2-tier scope note.
278
397
 
279
398
  ## 0.6.3 (2026-04-20)
280
399
 
@@ -283,7 +402,7 @@ end
283
402
  - **`runs:` parameter on `compare_models` and `optimize_retry_policy`** — runs each candidate N times per eval and aggregates the mean score, mean cost per run, and mean latency. Reduces sampling variance in live mode where LLM outputs are non-deterministic (gpt-5 family enforces `temperature=1.0` server-side, so a single unlucky sample can misclassify a viable candidate as "failing"). Default `runs: 1` — backward compatible.
284
403
  - **`RUNS=N` on `rake ruby_llm_contract:optimize`** — CLI flag for variance-aware optimization.
285
404
  - **`Eval::AggregatedReport`** — duck-type `Report` exposing `score` (mean), `score_min`/`score_max` (spread), `total_cost` (mean per run), `pass_rate` (clean-pass count x/N), and `clean_passes`.
286
- - **Guide: [Reducing variance with `runs:`](docs/guide/optimizing_retry_policy.md#reducing-variance-with-runs)** — when to use it and why.
405
+ - **Guide: [Reducing variance with `runs:`](docs/guide/optimizing_retry_policy.md)** — when to use it and why.
287
406
 
288
407
  ## 0.6.2 (2026-04-18)
289
408
 
@@ -305,9 +424,9 @@ end
305
424
 
306
425
  ### Features
307
426
 
308
- - **Multi-provider operator tooling** — rake tasks support `PROVIDER=openai|anthropic|ollama`, `CANDIDATES=model@effort,...`, and `REASONING_EFFORT=low|medium|high`.
427
+ - **Multi-provider operator tooling** — rake tasks support `PROVIDER=openai|anthropic|ollama` and `CANDIDATES=model@effort,...`.
309
428
  - **`rake ruby_llm_contract:recommend`** — wraps `Step.recommend` with CLI interface, prints best config, retry chain, DSL, rationale, and savings.
310
- - **Ollama support** — `PROVIDER=ollama` with configurable `OLLAMA_API_BASE`.
429
+ - **Ollama support** — `PROVIDER=ollama` (base URL via `RubyLLM.configure { |c| c.ollama_api_base = ... }`).
311
430
 
312
431
  ## 0.6.0 (2026-04-12)
313
432
 
data/README.md CHANGED
@@ -23,7 +23,7 @@ end
23
23
  RubyLLM::Contract.configure { }
24
24
  ```
25
25
 
26
- Works with any `ruby_llm` provider (OpenAI, Anthropic, Gemini, etc). Requires `ruby_llm ~> 1.12` and Ruby ≥ 3.2.
26
+ Works with any `ruby_llm` provider (OpenAI, Anthropic, Gemini, etc). Requires `ruby_llm ~> 2.0` and Ruby ≥ 3.2.
27
27
 
28
28
  ## Example
29
29
 
@@ -132,6 +132,8 @@ Everything below is optional — the example above is a complete step. Reach for
132
132
 
133
133
  Also supports [multi-step pipelines](docs/guide/pipeline.md) with fail-fast and per-step models.
134
134
 
135
+ **Runnable companion repo:** [`ruby_llm-contract_demo`](https://github.com/justi/ruby_llm-contract_demo) — full LLM-as-judge lifecycle (v1 strict → v2 drift → judge calibration → v3/v4 iteration) on a returns-policy chatbot scenario, dual-language (EN default, `DEMO_LANG=pl` opt-in). Live OpenAI calls (`gpt-4.1-mini`, `temperature 0`), `LIVE=1` opt-in, ~$0.30 for the full lifecycle. See [llm_judge.md](docs/guide/llm_judge.md) for the methodology.
136
+
135
137
  ## Relation to `RubyLLM::Agent`
136
138
 
137
139
  `Step::Base` and `RubyLLM::Agent` (since RubyLLM 1.12) are **siblings** targeting the same niche: reusable, class-based prompts. Both call into `RubyLLM::Chat` directly — Step does not wrap Agent. Step adds the contract layer: `validate` (business invariants), `retry_policy escalate(...)` (model escalation on validation failure), `max_cost` pre-flight refusal, regression-eval framework, pipeline composition. **[Full feature mapping →](docs/guide/relation_to_agent.md)**
@@ -162,10 +164,18 @@ Different layers, complementary. [`ruby_llm-tribunal`](https://github.com/Alqemi
162
164
  | [Multimodal input (PDF / image / audio)](docs/guide/multimodal_input.md) | Route attachments through the contract; `attachment_token_estimate`, fail-closed cost, calibration table |
163
165
  | [Schema DSL reference](docs/guide/output_schema.md) | Every constraint, nested objects, pattern table |
164
166
  | [Prompt DSL reference](docs/guide/prompt_ast.md) | `system` / `rule` / `section` / `example` / `user` nodes |
167
+ | [Architecture overview](docs/architecture.md) | Class map: Pipeline, Step, Contract, Prompt, Eval |
165
168
 
166
169
  ## Status & versioning
167
170
 
168
- Pre-1.0 (currently **0.10.4**). Semver tracked; breaking changes flagged in [CHANGELOG](CHANGELOG.md). Pin `~> 0.10.4` until 1.0 ships.
171
+ Stable at **1.0.0**. Semver tracked; breaking changes flagged in [CHANGELOG](CHANGELOG.md). Pin `~> 1.0`.
172
+
173
+ What 1.0 promises not to break inside the 1.x line: the documented DSL
174
+ (`prompt`, `output_schema`, `validate`, `retry_policy`, `max_cost`, `define_eval`,
175
+ pipelines), the `Result` and `trace` shapes including `trace[:usage]`'s
176
+ `{ input_tokens:, output_tokens: }` keys, and the adapter interface. A new major
177
+ version of a runtime dependency may require a new major version here — that is
178
+ what 1.0.0 itself was.
169
179
 
170
180
  ## FAQ
171
181
 
@@ -120,7 +120,7 @@ report = AccuracyJudge.run_eval("calibration")
120
120
  report.score # => 0.92 → judge agrees with humans 92% of the time
121
121
  ```
122
122
 
123
- A reasonable bar is `score >= 0.85` — an empirical baseline from LLM-eval field practice ([Eugene Yan on evals](https://eugeneyan.com/writing/evals/), [Hamel Husain field guide](https://hamel.dev/blog/posts/field-guide/)). Below that, judge-human agreement drops into a band where the judge's confident percentages stop matching what a human reviewer would say, and gate decisions become unreliable. If your domain is high-stakes (medical, legal, financial), raise the bar to 0.90+. Iterate the judge's prompt, or escalate to a stronger model, until the score on real production data crosses your bar.
123
+ A reasonable bar is `score >= 0.85` — a practical threshold synthesized from field practice. [Hamel Husain](https://hamel.dev/blog/posts/field-guide/) reports it took **three iterations** of the judge prompt to reach ">90% agreement" with humans in a Honeycomb case study; [Eugene Yan](https://eugeneyan.com/writing/evals/) notes RAG-grounded systems still hold a 5-10% factual inconsistency rate even after good prompt engineering. The specific 0.85 number is the author's summary — neither source quotes it directly. Below ~0.85, judge-human agreement drops into a band where the judge's confident percentages stop matching what a human reviewer would say, and gate decisions become unreliable. If your domain is high-stakes (medical, legal, financial), raise the bar to 0.90+. Iterate the judge's prompt, or escalate to a stronger model, until the score on real production data crosses your bar.
124
124
 
125
125
  **What "iterate the judge's prompt" actually looks like.** The first judge prompt almost always over-flags. Two or three iterations are normal before the judge is ready to gate anything. The common pattern:
126
126
 
@@ -195,9 +195,9 @@ What to do operationally:
195
195
 
196
196
  The drop is your signal to refine the judge's prompt, not to lower the gate.
197
197
 
198
- ## When to escalate to Tribunal's catalog
198
+ ## When to reach for Tribunal instead
199
199
 
200
- If you find yourself building three or four judges that all rhyme — "faithful?", "hallucination?", "refusal?", "PII leakage?" — you are reinventing what [`ruby_llm-tribunal`](https://github.com/Alqemist-labs/ruby_llm-tribunal) ships as a built-in catalog. See [Relation to Tribunal](relation_to_tribunal.md) for the integration recipe: Contract Steps make the LLM calls, Tribunal supplies the grading vocabulary, and the two compose in a single `define_eval`.
200
+ [`ruby_llm-tribunal`](https://github.com/Alqemist-labs/ruby_llm-tribunal) ships an off-the-shelf catalog of common LLM-as-judge assertions (`assert_faithful`, `assert_hallucination`, `assert_refusal`, `assert_no_pii`, etc.) — a shortcut when your check matches one of those domain-general categories. The methodology in this guide still applies: calibrate the judge against your human-labeled production data **before** trusting Tribunal's `default_threshold = 0.8`, refine the prompt when it over-flags, watch for the anti-patterns above. See [Relation to Tribunal](relation_to_tribunal.md) for the full positioning — what each gem documents (and doesn't), a concrete decision tree on catalog-vs-custom-judge, and three working integration patterns.
201
201
 
202
202
  ## See also
203
203
 
@@ -2,7 +2,7 @@
2
2
 
3
3
  > Read this when your contract needs to send a PDF, image, or audio file to the LLM — not just text.
4
4
 
5
- `ruby_llm-contract` 0.10.0+ routes attachments through the contract layer, so `max_cost`, `validate`, `retry_policy escalate(...)`, and trace observability still apply. The gem does **not** ship its own multimodal API — it forwards `with:` to `RubyLLM::Chat#ask`, which RubyLLM 1.15+ normalises per provider (Anthropic, OpenAI, Gemini).
5
+ `ruby_llm-contract` 0.10.0+ routes attachments through the contract layer, so `max_cost`, `validate`, `retry_policy escalate(...)`, and trace observability still apply. The gem does **not** ship its own multimodal API — it forwards `with:` to `RubyLLM::Chat#ask`, which RubyLLM normalises per provider (Anthropic, OpenAI, Gemini).
6
6
 
7
7
  ## Minimal example
8
8
 
@@ -130,7 +130,7 @@ Pick a value at or above the provider's worst-case. The estimate is a **floor fo
130
130
 
131
131
  ## Multi-turn caveat
132
132
 
133
- If your contract uses history (`add_history`), attachments from prior turns are **not** replayed in 0.10.x. Single-turn multimodal works; follow-up questions on the same document require additional work that is deferred to a later release. See [ADR-0022](../decisions/ADR-0022-v09-multimodal-input.md) (internal) for the rationale.
133
+ If your contract uses history (`add_history`), attachments from prior turns are **not** replayed (still true in 1.0.x). Single-turn multimodal works; follow-up questions on the same document require additional work that is deferred to a later release. The rationale is recorded in the project's internal ADR-0022, which is not shipped with the gem.
134
134
 
135
135
  ## Provider notes
136
136
 
@@ -149,7 +149,8 @@ RSpec.describe ExtractInvoiceData do
149
149
  it "forwards attachment to chat.ask" do
150
150
  expect_any_instance_of(RubyLLM::Chat).to receive(:ask)
151
151
  .with(anything, with: "fixtures/invoice.pdf")
152
- .and_return(double(content: '{"vendor":"X",...}', input_tokens: 200, output_tokens: 50))
152
+ .and_return(double(content: '{"vendor":"X",...}',
153
+ tokens: RubyLLM::Tokens.new(input: 200, output: 50)))
153
154
 
154
155
  result = described_class.run("extract", context: { attachment: "fixtures/invoice.pdf" })
155
156
  expect(result.status).to eq(:ok)
@@ -2,12 +2,12 @@
2
2
 
3
3
  > Read this as a reference for the schema DSL — every constraint, nested objects, arrays of objects, the full pattern table.
4
4
 
5
- Declare the expected output structure using [ruby_llm-schema](https://github.com/danielfriis/ruby_llm-schema) DSL. The schema serves **two purposes**:
5
+ Declare the expected output structure using the [schematist](https://github.com/crmne/schematist) DSL. The schema serves **two purposes**:
6
6
 
7
7
  1. **Output validation** — replaces type and shape checks (enums, ranges, required fields). One declaration instead of many.
8
8
  2. **Provider-side request** — with the RubyLLM adapter, the schema is sent to the LLM provider via `chat.with_schema(...)`, asking the model to return JSON matching the shape. Cheaper models sometimes ignore the request, which is why client-side validation (point 1) still matters.
9
9
 
10
- > **Same DSL `RubyLLM::Agent.schema` accepts.** Both use `RubyLLM::Schema.create(&block)` under the hood. `Step.output_schema` is eager-compiled at class load and drives client-side validation; `Agent.schema` accepts a `Proc` for dynamic per-call schemas. See [Relation to Agent](relation_to_agent.md) for the full comparison.
10
+ > **Same DSL `RubyLLM::Agent.schema` accepts.** Both use `Schematist::Schema.create(&block)` under the hood (schematist is what the `ruby_llm-schema` gem became; RubyLLM 2.x depends on it directly). `Step.output_schema` is eager-compiled at class load and drives client-side validation; `Agent.schema` accepts a `Proc` for dynamic per-call schemas. See [Relation to Agent](relation_to_agent.md) for the full comparison.
11
11
 
12
12
  All examples below extend the `SummarizeArticle` step from the [README](../../README.md).
13
13
 
@@ -73,6 +73,34 @@ Tribunal grades **a fixed set of cases on every PR** to catch quality regression
73
73
 
74
74
  **Both.** You ship contracts in prod (Contract) AND want stronger CI signal beyond schema regression — judge-quality grading on a frozen dataset, plus adversarial red-team probes. Use Contract's `Step` to make the call, run it in `define_eval` over your dataset, and grade each case with Tribunal helpers in your spec or via the dataset's `evaluator:` proc.
75
75
 
76
+ ### Tribunal's catalog vs Contract's `llm_judge.md` — concrete decision tree
77
+
78
+ If you specifically need an **LLM-as-judge** (a second LLM grading the first one's output), the decision is:
79
+
80
+ - **Reach for Tribunal's catalog** when your check is one of the well-defined, domain-general categories Tribunal ships: *"is this faithful to the retrieved context?"*, *"is this a refusal?"*, *"does this contain PII?"*, *"is this jailbreak-resistant?"*, *"hallucinated?"*, *"toxic?"*, *"biased?"*. One line in a spec, default threshold, no judge code to write or maintain. The judge prompt is baked into the gem.
81
+ - **Build a custom judge per [`llm_judge.md`](llm_judge.md)** when:
82
+ - **Your criterion is domain-specific** — *"does this medical advice match our internal safety policy?"*, *"is this reply in our brand voice?"*, *"does this summary preserve the legal disclaimer verbatim?"*. No off-the-shelf judge knows your policy; you write the prompt.
83
+ - **You need the verdict inside a `define_eval` regression gate** (the `evaluator:` lambda pattern) — Tribunal's surface is spec-time assertions, not eval-framework evaluators.
84
+ - **You need a per-claim breakdown** (sentence-level *"this claim → unsupported, that claim → contradicted"* output) for PR debugging — Tribunal returns one score per assertion.
85
+ - **You need to iterate the judge prompt** because it over-flags on your data — Tribunal's prompts are fixed per assertion.
86
+ - **Use both** for the same project even when your check is in Tribunal's catalog: Tribunal's `assert_faithful` for spec-time grade on individual responses, plus a calibrated custom judge wired as `evaluator:` in a regression `define_eval` over a frozen dataset for CI merge-gating. They cover different lifecycle stages.
87
+
88
+ Either way, the **methodology** in [`llm_judge.md`](llm_judge.md) — calibrate the judge against human-labeled production samples before trusting any score, watch for the six anti-patterns, refine the prompt when it over-flags — applies equally to Tribunal's built-ins, Tribunal's custom registered judges, and Contract `Step::Base` judges. Tribunal's `default_threshold = 0.8` is a starting point, not a calibrated bar for your data.
89
+
90
+ ## What Tribunal documents — and what it doesn't
91
+
92
+ Tribunal's README ships an **implementation catalog** (`assert_faithful`, `assert_hallucination`, `assert_refusal`, `assert_no_pii`, `assert_no_toxicity`, `assert_no_bias`, `assert_jailbreak_resistant`, etc., plus a `register_judge` API for custom ones). What it currently leaves to the adopter:
93
+
94
+ | Tribunal ships | Tribunal's README doesn't document (Contract's [`llm_judge.md`](llm_judge.md) does) |
95
+ |---|---|
96
+ | `default_threshold = 0.8` (fixed) | How to **calibrate** the threshold against your human-labeled production data |
97
+ | Judge prompt baked in per assertion | How to **iterate the judge prompt** when it over-flags stylistic courtesy as drift |
98
+ | Single score per assertion | **Per-claim breakdown** schema for sentence-level PR debugging |
99
+ | `assert_faithful` in a spec | **Judge as `evaluator:` lambda** in a Contract `define_eval` regression gate |
100
+ | Custom Judge mechanism (`register_judge`) | **Anti-patterns** (stubbing the verdict, calibrating on synthetic data, calibrating once and shipping) |
101
+
102
+ This is a **complementary gap**, not a competition. Tribunal owns the implementation catalog; Contract's `llm_judge.md` owns the methodology. A typical production setup uses both layers: pick (or build) the implementation, then calibrate it against your humans **before** trusting any score.
103
+
76
104
  ## Integration patterns
77
105
 
78
106
  These work today without any code changes in either gem — both use plain Ruby blocks/procs as extension points.
data/examples/README.md CHANGED
@@ -14,7 +14,7 @@ Pedagogical order: hook → activation → evolution → composition → quality
14
14
  | 05 | `05_eval_dataset.rb` | **"How do I stop silent prompt regressions?"** — define_eval with real cases, baseline vs regressed adapter, regression detection signal, inline eval_case. |
15
15
  | 06 | `06_retry_variants.rb` | **"What retry shapes exist beyond cross-model?"** — `attempts: 3` (variance absorption), `reasoning_effort` escalation (low→medium→high), cross-provider fallback (Ollama → Anthropic → OpenAI). |
16
16
 
17
- Every example has an "Expected output" section in the file header — you can read what each one prints without running it.
17
+ Four of the seven carry an "Expected output" section in the file header (01, 03, 05, 06), so you can read what they print without running them. The other three differ: `00_basics.rb` prints nothing and annotates each result inline with `# =>`; `02_real_llm_minimal.rb` needs a live provider key, so its output is not deterministic; `04_summarize_and_translate.rb` prints but has no header block.
18
18
 
19
19
  ## Running
20
20
 
@@ -25,7 +25,6 @@ module RubyLLM
25
25
  build_response(response)
26
26
  end
27
27
 
28
- # Maps option keys to the RubyLLM chat method and argument form.
29
28
  CHAT_OPTION_METHODS = {
30
29
  temperature: :with_temperature,
31
30
  schema: :with_schema
@@ -70,12 +69,10 @@ module RubyLLM
70
69
  thinking_config = resolve_thinking_config(options)
71
70
  chat.with_thinking(**thinking_config) if thinking_config
72
71
 
73
- # `with_params` carries only raw passthroughs (currently `max_tokens`).
74
- # `reasoning_effort` is no longer forwarded here — it goes through
75
- # `with_thinking` above, which is the canonical RubyLLM API.
76
- params = {}
77
- params[:max_tokens] = options[:max_tokens] if options[:max_tokens]
78
- chat.with_params(**params) if params.any?
72
+ # RubyLLM 2.0 removed `with_params`; `with_max_output_tokens` is the
73
+ # replacement for this one passthrough. `reasoning_effort` is not
74
+ # forwarded here — it goes through `with_thinking` above.
75
+ chat.with_max_output_tokens(options[:max_tokens]) if options[:max_tokens]
79
76
  end
80
77
 
81
78
  # Returns merged `{ effort:, budget: }` or nil. `options[:reasoning_effort]`
@@ -91,11 +88,18 @@ module RubyLLM
91
88
  content = response.content
92
89
  content = content.to_s unless content.is_a?(Hash) || content.is_a?(Array)
93
90
 
91
+ # This is the ONLY place upstream token counts are read. RubyLLM 2.0
92
+ # replaced `Message#input_tokens`/`#output_tokens` with a `Tokens` value
93
+ # object. The `{ input_tokens:, output_tokens: }` shape below is this
94
+ # gem's own public contract (Step::Trace#usage, cost calculation,
95
+ # documented in the README) and deliberately does NOT change.
96
+ tokens = response.tokens
97
+
94
98
  Response.new(
95
99
  content: content,
96
100
  usage: {
97
- input_tokens: response.input_tokens || 0,
98
- output_tokens: response.output_tokens || 0
101
+ input_tokens: tokens&.input || 0,
102
+ output_tokens: tokens&.output || 0
99
103
  }
100
104
  )
101
105
  end
@@ -3,8 +3,6 @@
3
3
  module RubyLLM
4
4
  module Contract
5
5
  module Concerns
6
- # Shared helpers for context hash manipulation.
7
- # Used by EvalHost, Runner, Step::Base.
8
6
  module ContextHelpers
9
7
  private
10
8
 
@@ -3,7 +3,6 @@
3
3
  module RubyLLM
4
4
  module Contract
5
5
  module Concerns
6
- # Recursively converts Hash keys to symbols while preserving array shape.
7
6
  module DeepSymbolize
8
7
  def deep_symbolize(object)
9
8
  case object
@@ -38,7 +38,6 @@ module RubyLLM
38
38
  end
39
39
  private_class_method :parse_json_text
40
40
 
41
- # Fallback: attempt to extract the first JSON object or array from prose
42
41
  def self.parse_json_with_extraction(text, raw_output)
43
42
  extracted = extract_json(text)
44
43
  unless extracted
@@ -72,8 +71,6 @@ module RubyLLM
72
71
  match ? match[1] : text
73
72
  end
74
73
 
75
- # Extract the first JSON object or array from text that may contain prose.
76
- # Uses bracket-matching to find the outermost balanced { } or [ ] block.
77
74
  JSON_START_PATTERN = /[{\[]/
78
75
 
79
76
  def self.extract_json(text)
@@ -3,7 +3,6 @@
3
3
  module RubyLLM
4
4
  module Contract
5
5
  class SchemaValidator
6
- # Validates numeric and collection size bounds for a node.
7
6
  class BoundRule
8
7
  NUMERIC_LIMITS = [
9
8
  { key: :minimum, label: "minimum", relation: "below", invalid: ->(actual, limit) { actual < limit } },
@@ -3,7 +3,6 @@
3
3
  module RubyLLM
4
4
  module Contract
5
5
  class SchemaValidator
6
- # Validates enum membership for a node when enum values are declared.
7
6
  class EnumRule
8
7
  def initialize(errors)
9
8
  @errors = errors
@@ -21,10 +21,6 @@ module RubyLLM
21
21
  value.is_a?(Array)
22
22
  end
23
23
 
24
- def numeric?
25
- value.is_a?(Numeric)
26
- end
27
-
28
24
  def properties
29
25
  schema[:properties] || {}
30
26
  end
@@ -3,7 +3,6 @@
3
3
  module RubyLLM
4
4
  module Contract
5
5
  class SchemaValidator
6
- # Applies scalar-only validation rules to a schema node.
7
6
  class ScalarRules
8
7
  def initialize(errors)
9
8
  @rules = [
@@ -3,7 +3,6 @@
3
3
  module RubyLLM
4
4
  module Contract
5
5
  class SchemaValidator
6
- # Validates the declared JSON schema type for a node.
7
6
  class TypeRule
8
7
  def initialize(errors)
9
8
  @errors = errors
@@ -84,11 +84,36 @@ module RubyLLM
84
84
  end
85
85
 
86
86
  def self.compute_cost(model_info, usage)
87
- input_cost = token_cost(usage[:input_tokens], model_info.input_price_per_million)
88
- output_cost = token_cost(usage[:output_tokens], model_info.output_price_per_million)
87
+ prices = prices_for(model_info)
88
+ return nil unless prices
89
+
90
+ input_cost = token_cost(usage[:input_tokens], prices[:input])
91
+ output_cost = token_cost(usage[:output_tokens], prices[:output])
89
92
  (input_cost + output_cost).round(6)
90
93
  end
91
94
 
95
+ # Two shapes reach here. `RegisteredModel` (our own struct, from
96
+ # `register_model`) exposes flat `*_price_per_million` readers. RubyLLM 2.0
97
+ # moved provider pricing into a nested value object and dropped those flat
98
+ # readers, so it has to be walked: pricing -> text_tokens -> standard.
99
+ #
100
+ # nil means "unknown pricing", which `max_cost` fails closed on. The guards
101
+ # keep the two known shapes explicit, but `calculate` rescues StandardError,
102
+ # so a future shape this walk cannot read still degrades to nil rather than
103
+ # surfacing - every budgeted call would be refused with no error.
104
+ def self.prices_for(model_info)
105
+ if model_info.respond_to?(:input_price_per_million)
106
+ return { input: model_info.input_price_per_million,
107
+ output: model_info.output_price_per_million }
108
+ end
109
+
110
+ tier = model_info.pricing&.text_tokens&.standard if model_info.respond_to?(:pricing)
111
+ return unless tier.respond_to?(:input_per_million)
112
+
113
+ { input: tier.input_per_million, output: tier.output_per_million }
114
+ end
115
+ private_class_method :prices_for
116
+
92
117
  # Provider pricing is denominated per 1M tokens; divide here to get
93
118
  # the dollar cost for the actual usage count. Named constant for
94
119
  # consistency with how RubyLLM and provider docs express prices.
@@ -79,8 +79,10 @@ module RubyLLM
79
79
  private
80
80
 
81
81
  def compute_score(cases)
82
- # Exclude skipped cases from score (consistent with Report#score)
83
- evaluated = cases.reject { |c| c[:details]&.start_with?("skipped:") }
82
+ # Asymmetric on purpose: the baseline side arrives already filtered by
83
+ # ReportStats#evaluated_results, the current side does not, so this
84
+ # string check is the only skip filter on one of the two sides.
85
+ evaluated = cases.reject { |c| c[:details]&.start_with?(CaseResult::SKIPPED_DETAILS_PREFIX) }
84
86
  return 0.0 if evaluated.empty?
85
87
 
86
88
  evaluated.sum { |c| c[:score] } / evaluated.length
@@ -0,0 +1,31 @@
1
+ # frozen_string_literal: true
2
+
3
+ module RubyLLM
4
+ module Contract
5
+ module Eval
6
+ # The `model (effort: x)` notation, rendered into optimizer tables and parsed
7
+ # back to rebuild candidate configs. Render and parse live together because a
8
+ # format change on one side silently mis-parses on the other: the pattern is
9
+ # built from the same literal the renderer emits.
10
+ module CandidateLabel
11
+ OPEN = " (effort: "
12
+ CLOSE = ")"
13
+ PATTERN = /\s*#{Regexp.escape(OPEN.lstrip)}(\w+)#{Regexp.escape(CLOSE)}/
14
+
15
+ def self.render(config)
16
+ effort = config[:reasoning_effort]
17
+ return config[:model] unless effort
18
+
19
+ "#{config[:model]}#{OPEN}#{effort}#{CLOSE}"
20
+ end
21
+
22
+ def self.parse(label)
23
+ match = PATTERN.match(label)
24
+ return { model: label } unless match
25
+
26
+ { model: match.pre_match.strip, reasoning_effort: match[1] }
27
+ end
28
+ end
29
+ end
30
+ end
31
+ end
@@ -30,7 +30,7 @@ module RubyLLM
30
30
  private
31
31
 
32
32
  def missing_adapter?(error)
33
- error.message.include?("No adapter configured")
33
+ error.message.include?(Step::Base::NO_ADAPTER_MESSAGE)
34
34
  end
35
35
 
36
36
  def skipped_result(test_case, reason)
@@ -43,7 +43,7 @@ module RubyLLM
43
43
  score: 0.0,
44
44
  passed: false,
45
45
  label: "SKIP",
46
- details: "skipped: #{reason}"
46
+ details: CaseResult.skipped_details(reason)
47
47
  )
48
48
  end
49
49
  end
@@ -4,6 +4,14 @@ module RubyLLM
4
4
  module Contract
5
5
  module Eval
6
6
  class CaseResult
7
+ # BaselineDiff#compute_score keys off this prefix to keep skipped cases
8
+ # out of the score denominator. Both ends must read it from here.
9
+ SKIPPED_DETAILS_PREFIX = "skipped:"
10
+
11
+ def self.skipped_details(reason)
12
+ "#{SKIPPED_DETAILS_PREFIX} #{reason}"
13
+ end
14
+
7
15
  attr_reader :name, :input, :output, :expected, :step_status,
8
16
  :score, :details, :duration_ms, :cost, :attempts
9
17
 
@@ -3,8 +3,6 @@
3
3
  module RubyLLM
4
4
  module Contract
5
5
  module Eval
6
- # Extracted from Runner to reduce class length.
7
- # Builds contract detail strings for contract-only evaluation.
8
6
  module ContractDetailBuilder
9
7
  private
10
8
 
@@ -6,9 +6,10 @@ module RubyLLM
6
6
  class ModelComparison
7
7
  attr_reader :eval_name, :reports, :configs, :fallback
8
8
 
9
+ # Kept as the name six call sites already use; the format itself lives in
10
+ # CandidateLabel, which owns both directions.
9
11
  def self.candidate_label(config)
10
- effort = config[:reasoning_effort]
11
- effort ? "#{config[:model]} (effort: #{effort})" : config[:model]
12
+ CandidateLabel.render(config)
12
13
  end
13
14
 
14
15
  def initialize(eval_name:, reports:, configs: nil, fallback: nil)