activeagent 1.6.4 → 1.7.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +297 -0
- data/README.md +3 -0
- data/lib/active_agent/concerns/view.rb +23 -0
- data/lib/active_agent/evals/model_spec.rb +4 -0
- data/lib/active_agent/evals/publisher.rb +110 -14
- data/lib/active_agent/evals/runner.rb +2 -2
- data/lib/active_agent/evals/scenario.rb +3 -3
- data/lib/active_agent/evals/scenario_parser.rb +3 -3
- data/lib/active_agent/evals/suite.rb +8 -8
- data/lib/active_agent/evals.rb +1 -1
- data/lib/active_agent/generation_job.rb +15 -1
- data/lib/active_agent/providers/_base_provider.rb +10 -2
- data/lib/active_agent/providers/anthropic/transforms.rb +57 -4
- data/lib/active_agent/providers/open_ai/responses/transforms.rb +8 -0
- data/lib/active_agent/providers/ruby_llm_provider.rb +53 -3
- data/lib/active_agent/version.rb +1 -1
- metadata +4 -4
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: 14e5b0cec7ba170c446d1ed32ad5c9d0705ac8eae4fdc6e4e4a7d17bdae502a2
|
|
4
|
+
data.tar.gz: 28e12f0ef3727a97fc231e5f7ace15ad4cb020b379fc76c757c2206c89f2b2e6
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: 9cd766d568ab510b043e4902c1c5fa73ce877a268761168509f2772190ab0f2d6f259fe62e4654aa2a002d20228f876fce82c7c2f5f3e1018cd6cb01935aed27
|
|
7
|
+
data.tar.gz: 197cab65243e11561c7216f08e9b30369819e3ef73627d0f6e72092a7fc74c16a278638f5044e11f292149a1a39dbe11cab9593a41e7289049a3a90c6232a0a0
|
data/CHANGELOG.md
CHANGED
|
@@ -7,6 +7,303 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0
|
|
|
7
7
|
|
|
8
8
|
## [Unreleased]
|
|
9
9
|
|
|
10
|
+
## [1.7.1] - 2026-09-29
|
|
11
|
+
|
|
12
|
+
Releases `activeagent` and `actionagent` 1.7.1 from one tag. A patch on 1.7.0:
|
|
13
|
+
the Evaluations page picks the judge and compared models from the provider
|
|
14
|
+
catalogs, `GET /api/provider_models` lists the host's RubyLLM registry, and
|
|
15
|
+
the RubyLLM provider requires ruby_llm 1.x and sends tool calls and structured
|
|
16
|
+
output in the shape ruby_llm reads. No migrations.
|
|
17
|
+
|
|
18
|
+
### Added
|
|
19
|
+
|
|
20
|
+
- **Model pickers on the Evaluations page** (`actionagent`). The judge model
|
|
21
|
+
and the models to compare, on the new-evaluation form and on a scenario
|
|
22
|
+
suite, are chosen from type-ahead suggestions, and a model the suggestions
|
|
23
|
+
lack can still be typed. The compare fields hold one removable chip per
|
|
24
|
+
model and still submit the same comma-separated list.
|
|
25
|
+
- With scenarios, the suggestions are the catalogs the agent builder uses,
|
|
26
|
+
for the providers the owner's runs have credentials for. A model whose
|
|
27
|
+
name alone would run elsewhere is offered with its provider in front, as
|
|
28
|
+
`ModelSpec.parse` reads it: `ollama/llama3.2`, or
|
|
29
|
+
`openrouter/anthropic/claude-sonnet-4.5` for OpenRouter's copy of an
|
|
30
|
+
Anthropic model.
|
|
31
|
+
- Without scenarios, a compared model selects the generations recorded
|
|
32
|
+
under its name, usually the provider's dated id, so the suggestions are
|
|
33
|
+
the names the agent's generations were recorded under, from the new
|
|
34
|
+
`GET /api/agents/:id/recorded_models`. Adding or clearing the scenarios
|
|
35
|
+
renames the catalog models already chosen to match.
|
|
36
|
+
- The judge field suggests the models of the provider the judge runs on
|
|
37
|
+
and says which provider that is, or that the credentials deciding it
|
|
38
|
+
could not be read.
|
|
39
|
+
|
|
40
|
+
`GET /api/evaluations` reports that provider as `judge_provider` and the
|
|
41
|
+
providers runs can use as `model_providers`
|
|
42
|
+
(`AgentExecutionService.available_providers`). Runs and their judge use
|
|
43
|
+
the evaluated agent's owner's credentials, so both fields describe that
|
|
44
|
+
owner's when `agent_id` scopes the list, and the signed-in owner's
|
|
45
|
+
otherwise. `judge_provider` is null when no provider has credentials, and
|
|
46
|
+
also when reading them raised, which `judge_provider_error: true` marks.
|
|
47
|
+
`model_providers` leaves out a provider whose credentials cannot be read.
|
|
48
|
+
Either way the list still loads.
|
|
49
|
+
- **`GET /api/provider_models` lists the host's RubyLLM registry**
|
|
50
|
+
(`actionagent`). When the host app loads RubyLLM, the chat models its
|
|
51
|
+
registry lists for the provider that take and return text (RubyLLM's
|
|
52
|
+
bundled catalog, or the host's own model table) follow the live or curated
|
|
53
|
+
list, each once, so the builder's preselected default is unchanged. That
|
|
54
|
+
leaves out the speech, transcription, moderation and completion-only
|
|
55
|
+
models RubyLLM counts as chat models, which its registry lists with no
|
|
56
|
+
modalities or as taking audio, and with them the few chat models it lists
|
|
57
|
+
with no modalities, which can still be typed. A registry that raises is
|
|
58
|
+
logged and leaves the list as it was.
|
|
59
|
+
|
|
60
|
+
### Changed
|
|
61
|
+
|
|
62
|
+
- **The RubyLLM provider requires ruby_llm 1.x** (`activeagent`). ruby_llm
|
|
63
|
+
2.0 renamed the APIs `RubyLLMProvider` calls, and the open `>= 1.0`
|
|
64
|
+
requirement let `bundle update` install it. The provider now requires
|
|
65
|
+
`~> 1.0` until it supports 2.0 (#502). Loading it with an unsupported
|
|
66
|
+
version names the supported range and the loaded version, instead of
|
|
67
|
+
asking for a gem that is already in the Gemfile. Pin
|
|
68
|
+
`gem "ruby_llm", "~> 1.0"` if your bundle resolved 2.0.
|
|
69
|
+
|
|
70
|
+
### Fixed
|
|
71
|
+
|
|
72
|
+
- **Tool calls sent back through the RubyLLM provider** (`activeagent`).
|
|
73
|
+
After a tool ran, the follow-up request repeated the model's tool call
|
|
74
|
+
with its arguments as a JSON string where ruby_llm expects a Hash: OpenAI
|
|
75
|
+
received them JSON-encoded twice, and Anthropic received a string for
|
|
76
|
+
`tool_use.input`, which its API requires to be an object. The same
|
|
77
|
+
happened when a stored conversation containing a tool call was replayed.
|
|
78
|
+
The provider now hands ruby_llm the parsed arguments (#501).
|
|
79
|
+
- **Structured output through the RubyLLM provider** (`activeagent`). A
|
|
80
|
+
`json_schema` response_format reached ruby_llm unchanged, but ruby_llm
|
|
81
|
+
reads `{ name:, schema:, strict: }`, so OpenAI received a schema with a
|
|
82
|
+
null name and body, and the Anthropic request raised inside ruby_llm
|
|
83
|
+
before it was sent. The provider now converts it, naming the schema
|
|
84
|
+
`response` and making it strict unless the format says otherwise, as
|
|
85
|
+
ruby_llm's own `with_schema` does. A `text` format asks for plain text;
|
|
86
|
+
`json_object`, which ruby_llm has no mode for, and a `json_schema`
|
|
87
|
+
without a schema now raise `ArgumentError` instead of sending a request
|
|
88
|
+
the API rejects (#501).
|
|
89
|
+
|
|
90
|
+
## [1.7.0] - 2026-09-24
|
|
91
|
+
|
|
92
|
+
Releases `activeagent` and `actionagent` 1.7.0 from one tag. A minor release:
|
|
93
|
+
a mounted engine collects the evaluation reports applications publish with
|
|
94
|
+
`ActiveAgent::Evals::Publisher`, whose failures now say what the collector
|
|
95
|
+
refused and whether to retry; observed agents read their own traces in
|
|
96
|
+
evaluation criteria, the Tools and Traces tabs and deploy markers; host apps
|
|
97
|
+
extend the engine's models and controllers through concerns and mirror their
|
|
98
|
+
agent classes into the dashboard; and the Evaluations page is rebuilt around
|
|
99
|
+
runs. Run the install generator after upgrading (see Upgrading below).
|
|
100
|
+
|
|
101
|
+
Upgrading: the install generator emits two new migrations, both guarded
|
|
102
|
+
column by column: `ensure_agent_release_columns`, which adds the agent release
|
|
103
|
+
columns an install generated fresh on 1.6.2-1.6.4 never got (and those on
|
|
104
|
+
tables with a custom `table_name_prefix`), and `add_evaluation_report_identity`
|
|
105
|
+
for the evaluation report collector. Re-run
|
|
106
|
+
`bin/rails generate action_agent:install --skip` (`--skip` keeps your
|
|
107
|
+
initializer) and `bin/rails db:migrate`. A traces-only install needs neither;
|
|
108
|
+
if you re-run the generator there, pass `--traces_only` again, or it emits the
|
|
109
|
+
whole dashboard schema. Nothing else changes until an application publishes a
|
|
110
|
+
report to the mount.
|
|
111
|
+
|
|
112
|
+
### Added
|
|
113
|
+
|
|
114
|
+
- **Host concerns for the engine's models and controllers** (`actionagent`).
|
|
115
|
+
`ActionAgent.model_concerns` is included into
|
|
116
|
+
`ActionAgent::ApplicationRecord` as it loads, and so into every engine
|
|
117
|
+
model; `ActionAgent.controller_concerns` into
|
|
118
|
+
`ActionAgent::ApplicationController`, ahead of its own callbacks, and so
|
|
119
|
+
into every dashboard controller. (The ingest endpoint,
|
|
120
|
+
`Api::TracesController`, inherits `ActionController::API` and keeps its
|
|
121
|
+
own bearer-token authentication; it is not touched.) Entries are modules
|
|
122
|
+
or their names, resolved
|
|
123
|
+
when the class loads. A host that pins the engine's tables to one database
|
|
124
|
+
connection, or carries its session helpers onto the dashboard's
|
|
125
|
+
controllers, configures that here instead of reopening the classes from a
|
|
126
|
+
`to_prepare` block.
|
|
127
|
+
- **The Evaluations page is rebuilt around runs** (`actionagent`). Evaluations
|
|
128
|
+
are the top level; every run is kept and listed with its movement against
|
|
129
|
+
the run before it (`+3 passed vs #2`, `partial run`, `#1 failed`), and a
|
|
130
|
+
sampling evaluation's run opens to a page of its own at
|
|
131
|
+
`<mount>/evaluations/:id/runs/:run_id` — a scorecard per model cohort, the
|
|
132
|
+
judge's verdict, the criteria × models matrix and what the run asks to fix.
|
|
133
|
+
A scenario suite's runs are the same full-width list; a row selects the run
|
|
134
|
+
the suite's model scorecards, fix items and scenario matrix show.
|
|
135
|
+
- **What a run cost is two figures, not one.** The agent's spend — what the
|
|
136
|
+
replayed or sampled interactions cost to serve, with a `per_interaction`
|
|
137
|
+
rate, the operating cost a per-conversation budget is set against — is
|
|
138
|
+
reported apart from the judge's, the judge model's own calls, which run
|
|
139
|
+
agent-to-agent and offline. Every judge call is metered under what it was
|
|
140
|
+
for (`scores["_judge_usage"]`: calls, tokens, estimated cost and how many
|
|
141
|
+
calls scored, recommended, ruled or authored KPIs), and `EvaluationRun#usage`
|
|
142
|
+
carries both sides. The page shows them on every run row, on the run, on a
|
|
143
|
+
page tile and in the footer, so the cost of operating an agent is never
|
|
144
|
+
inflated by the cost of checking it.
|
|
145
|
+
- A generation-sampling run records `scores["_cohorts"]`: per model, how many
|
|
146
|
+
generations were sampled, how many cleared every criterion, their latency
|
|
147
|
+
and tokens, and what those interactions cost to serve.
|
|
148
|
+
- `GET /api/evaluations` carries `run_count` and a `previous_run` summary per
|
|
149
|
+
evaluation, and every serialized run its `number` in the evaluation's
|
|
150
|
+
history, oldest first.
|
|
151
|
+
- The dashboard's object lists hold their metric columns in place: a trace,
|
|
152
|
+
interaction or evaluation run with nothing in a column prints a dash there
|
|
153
|
+
rather than sliding its neighbours over (`MetaStrip`).
|
|
154
|
+
- `ActiveAgent::Base.rendered_instructions` renders an agent's instructions
|
|
155
|
+
outside a generation, for a dashboard mirroring the class and for tests
|
|
156
|
+
asserting what a model is told. Both otherwise reached a private renderer
|
|
157
|
+
through `send`.
|
|
158
|
+
- `ActionAgent::AgentSync` mirrors host agent classes into dashboard `Agent`
|
|
159
|
+
records, setting the `agent_class_name` that `AgentRelease` already expects a
|
|
160
|
+
host to have written. The code owns what an agent is (name, description,
|
|
161
|
+
instructions, tools — rewritten every sync); the operator owns how it runs
|
|
162
|
+
(provider, model, status — set on create and preserved), so a model chosen in
|
|
163
|
+
the dashboard survives the next deploy.
|
|
164
|
+
- `ActionAgent.run_host_agent_classes` (default `false`) runs an agent that
|
|
165
|
+
mirrors a host class as that class, rather than as one rebuilt from the
|
|
166
|
+
record's `tools` and `instructions` columns. Dashboard-authored agents, which
|
|
167
|
+
name no class, keep using the dynamic runtime either way; a class name that no
|
|
168
|
+
longer resolves falls back to it rather than failing the run.
|
|
169
|
+
- MCP servers take `allowed_tools` and `require_approval` in the common
|
|
170
|
+
format. OpenAI's Responses API receives both as given; Anthropic receives
|
|
171
|
+
`allowed_tools` as an `mcp_toolset` entry in `tools` (every other tool of
|
|
172
|
+
the server disabled), beside any tools the request already declares
|
|
173
|
+
(#328, by @dark-panda).
|
|
174
|
+
- **A mounted engine collects published evaluation reports** (`actionagent`).
|
|
175
|
+
An application that runs its agents itself and evaluates them in-process
|
|
176
|
+
publishes the finished report with `ActiveAgent::Evals::Publisher`; until
|
|
177
|
+
now a self-hosted install had nowhere to receive it. On a full install (not
|
|
178
|
+
one generated with `--traces_only`, which answers 501),
|
|
179
|
+
`POST <mount>/api/evaluation_reports` takes the version-1 envelope and
|
|
180
|
+
returns the receipt the publisher checks, and
|
|
181
|
+
`ActionAgent::EvaluationReportImport` stores it as the engine's own rows:
|
|
182
|
+
the observed agent for the report's `source` and `agent_name`, an evaluation
|
|
183
|
+
named for its suite and scope (`orders (eu, support)`), its scenarios, and a
|
|
184
|
+
complete run with a result per scenario and model. The Evaluations page
|
|
185
|
+
shows it the way it shows a run the dashboard executed, with the summary
|
|
186
|
+
recomputed from the stored results. The endpoint authenticates exactly as
|
|
187
|
+
trace ingest does (`ingest_api_key`, or the tenant's key in multi-tenant
|
|
188
|
+
mode), takes only `application/json`, and places a report's agent wherever
|
|
189
|
+
`trace_owner_resolver` puts that tenant's traced agents. A `run_id` is
|
|
190
|
+
stored once per tenant, or once per install, and compared exactly: 201 for a
|
|
191
|
+
new report, 200 for an identical retry, 409 for different content. Invalid
|
|
192
|
+
reports, and an evaluation name the report does not own, are 422; a cap an
|
|
193
|
+
operator has to lift (observed agents per owner, 100 evaluations per agent,
|
|
194
|
+
2,000 scenarios per evaluation) is 403; a new report over the new
|
|
195
|
+
`:evaluation_report` quota kind or past 30 new reports a minute from a key
|
|
196
|
+
is 429. An identical retry is never refused by the quota or the rate limit.
|
|
197
|
+
Bodies over 2 MiB are 413, and Rails never parses the body into params, so
|
|
198
|
+
nothing past the limit is read. `usage_recorder` is told
|
|
199
|
+
`:evaluation_report` for each stored report. Evaluation runs gain
|
|
200
|
+
`external_tenant`, `external_run_id` and `external_report_digest`, unique on
|
|
201
|
+
the first two (binary on MySQL); see Upgrading above.
|
|
202
|
+
`docs/evals/publication.md` documents the endpoint.
|
|
203
|
+
|
|
204
|
+
### Changed
|
|
205
|
+
|
|
206
|
+
- The engine's judge blocks take `ActiveAgent::Evals::Judge`'s `kind:`, so a
|
|
207
|
+
scenario run's score, recommendation and verdict calls are metered apart.
|
|
208
|
+
- `ActionAgent::TelemetryTrace` inherits `ActionAgent::ApplicationRecord`
|
|
209
|
+
like every other engine model (`actionagent`), so it carries the model
|
|
210
|
+
concerns above, `AdapterAware` and the ownership API (`owner_association`,
|
|
211
|
+
`for_owner`) from the same place. Its table name is unchanged.
|
|
212
|
+
- `Api::TracesController`'s bearer authentication and its 429 quota body
|
|
213
|
+
live in `ActionAgent::Api::IngestAuthentication` (`actionagent`), which the
|
|
214
|
+
evaluation report collector shares. A host subclass that overrides
|
|
215
|
+
`authenticate_api_key!` is unaffected. The tenant's
|
|
216
|
+
`increment_telemetry_usage!` is still called for each trace ingest request,
|
|
217
|
+
and not for a report post.
|
|
218
|
+
- `add_agent_releases` reads `ActionAgent.table_name_prefix` for the tables
|
|
219
|
+
it alters (`actionagent`), so a newly generated copy works on an install
|
|
220
|
+
with a custom prefix. The trace table keeps its fixed name.
|
|
221
|
+
- A collector's rejection of `ActiveAgent::Evals::Publisher` says what it
|
|
222
|
+
refused and whether to retry. The message carries the `error` string of a
|
|
223
|
+
JSON object response body beside the HTTP status — control characters and
|
|
224
|
+
runs of whitespace collapsed to one space, the API key replaced with
|
|
225
|
+
`[FILTERED]`, cut to 200 characters; nothing else from the body — and what
|
|
226
|
+
to do next: never retry the report under the same `run_id` on a 409,
|
|
227
|
+
publish a smaller selection on a 413, correct the report on a 422, retry
|
|
228
|
+
later on a 408, 429 or 5xx, and resolve the cause first on anything else,
|
|
229
|
+
such as a 401 or 403. `Publisher::Error` carries `status`, `detail` and
|
|
230
|
+
`retryable?`.
|
|
231
|
+
|
|
232
|
+
### Deprecated
|
|
233
|
+
|
|
234
|
+
- Assigning `ActionAgent.base_controller_class`, which has never been
|
|
235
|
+
consumed, warns through `ActionAgent.deprecator` and points at
|
|
236
|
+
`controller_concerns`. The accessor is removed in 2.0.
|
|
237
|
+
|
|
238
|
+
### Fixed
|
|
239
|
+
|
|
240
|
+
- Telemetry criteria (`trace_error_rate`, `trace_latency`) score an observed
|
|
241
|
+
agent from its own traces (`actionagent`). They selected traces by
|
|
242
|
+
`Agent#telemetry_agent_class`, which appends `Agent` to a class name
|
|
243
|
+
lacking it, so an agent observed from an application reporting `SupportBot`
|
|
244
|
+
found no traces and scored nothing, and observed agents of one class ending
|
|
245
|
+
in `Agent` read each other's actions. `Agent#telemetry_traces` selects the
|
|
246
|
+
traces `AgentRegistrar` attributed to the agent, plus unattributed ones with
|
|
247
|
+
its service, class and action. Deleting an observed agent leaves its traces
|
|
248
|
+
unattributed, so the agent registered again for them still reads them. The
|
|
249
|
+
agent's Traces tab, its Tools tab usage
|
|
250
|
+
columns and the Interactions list filtered to it use the same selection. The
|
|
251
|
+
Traces tab asks for it with `GET /api/traces?agent_id=`, which answers 404
|
|
252
|
+
for an agent the caller cannot see; `agent=` still filters by class. On the
|
|
253
|
+
Metrics page filtered to a class, an observed agent's deploy markers now
|
|
254
|
+
show under the class its traces report (`Agent#reported_agent_class`).
|
|
255
|
+
Authored and mirrored agents read the traces they did before.
|
|
256
|
+
- `Agent.prompt(...).generate_later` and `Agent.embed(...).embed_later` run
|
|
257
|
+
their job instead of raising `ArgumentError: unknown keywords` in the
|
|
258
|
+
worker (#346).
|
|
259
|
+
- The agent builder and editor can reach every model a provider serves: the
|
|
260
|
+
OpenRouter catalog is no longer cut to its first 100 ids, and the model
|
|
261
|
+
field is a type-ahead over the catalog that also takes an unlisted id
|
|
262
|
+
(`actionagent`, #427).
|
|
263
|
+
- A rejected Create Agent shows its validation errors on the builder — a
|
|
264
|
+
summary and a message under each field — instead of leaving the form
|
|
265
|
+
silently in place. `POST`/`PATCH /api/agents` 422s carry `field_errors`
|
|
266
|
+
beside `errors` (`actionagent`, #426).
|
|
267
|
+
- An engine agent is refused a provider whose client gem the host has not
|
|
268
|
+
installed (`openai` for OpenAI, Ollama and OpenRouter; `anthropic` for
|
|
269
|
+
Anthropic) when the provider is chosen, with a validation error naming
|
|
270
|
+
the gem, instead of failing on its first run (`actionagent`, #416).
|
|
271
|
+
- A fresh `action_agent:install` creates the agent release columns with the
|
|
272
|
+
dashboard tables (`actionagent`): `release_digest` on agents,
|
|
273
|
+
`release_digest` and `revision` on agent versions, and `agent_version_id`
|
|
274
|
+
on agent runs and evaluation runs. The generator emits `add_agent_releases`
|
|
275
|
+
before the create-table migration, so on a fresh install it found none of
|
|
276
|
+
those tables and added nothing, and creating an agent run or an evaluation
|
|
277
|
+
run raised `NoMethodError` on `agent_version_id`. An install generated
|
|
278
|
+
that way on 1.6.2-1.6.4 gets the columns from the new
|
|
279
|
+
`ensure_agent_release_columns` migration (see Upgrading above).
|
|
280
|
+
- Every failure of `ActiveAgent::Evals::Publisher` to deliver a report
|
|
281
|
+
raises `Publisher::Error`. A malformed response (`Net::HTTPBadResponse`,
|
|
282
|
+
`Net::HTTPHeaderSyntaxError`, or a `Zlib::Error` from corrupt compression)
|
|
283
|
+
escaped as its own class and is now a retryable `Publisher::Error`. A
|
|
284
|
+
report that cannot be encoded as JSON (invalid UTF-8, `NaN`, nesting too
|
|
285
|
+
deep) is now a non-retryable one raised before anything is sent: it escaped
|
|
286
|
+
as `JSON::GeneratorError`, or was blamed on the collector as invalid JSON.
|
|
287
|
+
Only a network failure keeps its underlying error as `cause`, so a response
|
|
288
|
+
body or report content never reaches a log through the exception chain.
|
|
289
|
+
Invalid arguments raise `ArgumentError`, now also for a `report` that does
|
|
290
|
+
not convert to a hash (a string raised `NoMethodError` and `nil` published
|
|
291
|
+
an empty report) and a `nil` timeout (`TypeError`).
|
|
292
|
+
- The publisher strips whitespace around its API key, so the key it sends is
|
|
293
|
+
the one it filters from a collector's explanation, and refuses a key with
|
|
294
|
+
characters other than visible ASCII.
|
|
295
|
+
|
|
296
|
+
### Security
|
|
297
|
+
|
|
298
|
+
- The dashboard's JSON API verifies the CSRF token (`actionagent`, #461). It
|
|
299
|
+
authenticates with the host's session cookie but had opted out of forgery
|
|
300
|
+
protection. The dashboard now sends the page's token with every mutating
|
|
301
|
+
request from one fetch shim; the MCP facade and trace ingest, which
|
|
302
|
+
authenticate by bearer token, stay exempt. A rejected request answers
|
|
303
|
+
`422` with `code: "invalid_csrf_token"`. Hosts that re-enabled protection
|
|
304
|
+
themselves (`ActionAgent::Api::BaseController.protect_from_forgery`) can
|
|
305
|
+
drop that line.
|
|
306
|
+
|
|
10
307
|
## [1.6.4] - 2026-09-22
|
|
11
308
|
|
|
12
309
|
Releases `activeagent` and `actionagent` 1.6.4 from one tag. A patch on 1.6.3
|
data/README.md
CHANGED
|
@@ -12,6 +12,29 @@ module ActiveAgent
|
|
|
12
12
|
include ActionView::Layouts
|
|
13
13
|
end
|
|
14
14
|
|
|
15
|
+
class_methods do
|
|
16
|
+
# The agent's rendered instructions, outside a generation.
|
|
17
|
+
#
|
|
18
|
+
# Two surfaces need the text an agent would run on without running it: a
|
|
19
|
+
# dashboard that mirrors the class (ActionAgent::AgentSync) and a test
|
|
20
|
+
# asserting what the model is told. Both otherwise reach a private
|
|
21
|
+
# renderer through `send`, which couples them to internals that can move
|
|
22
|
+
# without notice.
|
|
23
|
+
#
|
|
24
|
+
# TicketAgent.rendered_instructions # => "You are the Ticket agent..."
|
|
25
|
+
#
|
|
26
|
+
# TicketAgent.rendered_instructions(topic: "tickets")
|
|
27
|
+
#
|
|
28
|
+
# @param template [String] template name, default "instructions"
|
|
29
|
+
# @param assigns [Hash] instance variables the template reads
|
|
30
|
+
# @return [String, nil] nil when the agent has no such template
|
|
31
|
+
def rendered_instructions(template: "instructions", **assigns)
|
|
32
|
+
agent = new
|
|
33
|
+
assigns.each { |name, value| agent.instance_variable_set(:"@#{name}", value) }
|
|
34
|
+
agent.send(:view_render_template, template)
|
|
35
|
+
end
|
|
36
|
+
end
|
|
37
|
+
|
|
15
38
|
# Builds template lookup paths supporting both flat and nested directory structures.
|
|
16
39
|
#
|
|
17
40
|
# Templates are searched in priority order:
|
|
@@ -20,6 +20,10 @@ module ActiveAgent
|
|
|
20
20
|
class ModelSpec
|
|
21
21
|
DEFAULT_PROVIDERS = %w[openai anthropic ollama openrouter].freeze
|
|
22
22
|
|
|
23
|
+
# The dashboard's model pickers resolve names with a JavaScript copy of
|
|
24
|
+
# these rules and of .parse (actionagent/frontend/utils/modelOptions.mjs).
|
|
25
|
+
# actionagent/test/fixtures/model_spec_cases.json holds the cases both
|
|
26
|
+
# are tested against, so a change here needs the same change there.
|
|
23
27
|
DEFAULT_INFERENCE_RULES = [
|
|
24
28
|
[ /\Aclaude/i, "anthropic" ],
|
|
25
29
|
[ /\A(gpt-|o\d|chatgpt|text-embedding)/i, "openai" ],
|
|
@@ -4,6 +4,7 @@ require "json"
|
|
|
4
4
|
require "net/http"
|
|
5
5
|
require "openssl"
|
|
6
6
|
require "uri"
|
|
7
|
+
require "zlib"
|
|
7
8
|
|
|
8
9
|
module ActiveAgent
|
|
9
10
|
module Evals
|
|
@@ -11,11 +12,51 @@ module ActiveAgent
|
|
|
11
12
|
# retain run_id when retrying: compatible collectors treat that identity as
|
|
12
13
|
# immutable within the authenticated account. Delivery is blocking and does
|
|
13
14
|
# not follow redirects with the account's bearer credential.
|
|
15
|
+
#
|
|
16
|
+
# Every failure to deliver raises Error. Invalid arguments raise
|
|
17
|
+
# ArgumentError before anything is sent.
|
|
14
18
|
class Publisher
|
|
15
19
|
DEFAULT_ENDPOINT = "https://api.activeagents.ai/v1/evaluations"
|
|
16
20
|
MAX_BYTES = 2 * 1024 * 1024
|
|
17
|
-
|
|
21
|
+
DETAIL_LIMIT = 200
|
|
18
22
|
|
|
23
|
+
# Raised for every failed delivery. Only a network failure keeps the
|
|
24
|
+
# underlying error as its +cause+, so neither the response nor the
|
|
25
|
+
# report reaches a log through the exception chain.
|
|
26
|
+
#
|
|
27
|
+
# @!attribute [r] status
|
|
28
|
+
# @return [Integer, nil] the collector's HTTP status for a rejection, nil otherwise
|
|
29
|
+
# @!attribute [r] detail
|
|
30
|
+
# @return [String, nil] the collector's sanitized explanation of a rejection, if it gave one
|
|
31
|
+
class Error < StandardError
|
|
32
|
+
attr_reader :status, :detail
|
|
33
|
+
|
|
34
|
+
def initialize(message = nil, status: nil, detail: nil, retryable: false)
|
|
35
|
+
super(message)
|
|
36
|
+
@status = status
|
|
37
|
+
@detail = detail
|
|
38
|
+
@retryable = retryable
|
|
39
|
+
end
|
|
40
|
+
|
|
41
|
+
# Returns true when delivering the same report under the same run_id
|
|
42
|
+
# again may succeed.
|
|
43
|
+
def retryable?
|
|
44
|
+
@retryable
|
|
45
|
+
end
|
|
46
|
+
end
|
|
47
|
+
|
|
48
|
+
# Whether each rejection status is retryable, and what the caller should
|
|
49
|
+
# do about it. Other statuses fall back to the rules in +rejection+.
|
|
50
|
+
REJECTIONS = {
|
|
51
|
+
409 => [ false, "the collector already holds a different report under this run_id; never retry this report with the same run_id" ],
|
|
52
|
+
413 => [ false, "the report exceeds the collector's size limit; publish a smaller selection" ],
|
|
53
|
+
422 => [ false, "correct what the collector refused before retrying" ],
|
|
54
|
+
429 => [ true, "the account is over its quota or rate limit; retain the report and run_id and retry later" ]
|
|
55
|
+
}.freeze
|
|
56
|
+
|
|
57
|
+
# The key is sent as a bearer token and filtered from the collector's
|
|
58
|
+
# explanation, so it is limited to visible ASCII: the sanitizer in
|
|
59
|
+
# +collector_detail+ never alters it, and an echo of it always matches.
|
|
19
60
|
def initialize(api_key:, endpoint: DEFAULT_ENDPOINT, timeout: 10, open_timeout: 10)
|
|
20
61
|
@uri = URI.parse(endpoint.to_s)
|
|
21
62
|
unless @uri.is_a?(URI::HTTP) && @uri.host && !@uri.userinfo && !@uri.query && !@uri.fragment
|
|
@@ -24,12 +65,16 @@ module ActiveAgent
|
|
|
24
65
|
unless @uri.scheme == "https" || %w[localhost 127.0.0.1 ::1].include?(@uri.hostname)
|
|
25
66
|
raise ArgumentError, "Evaluation endpoint requires HTTPS except on loopback hosts"
|
|
26
67
|
end
|
|
27
|
-
raise ArgumentError, "Evaluation API key is required" if api_key.to_s.strip.empty?
|
|
28
68
|
|
|
29
|
-
@api_key = api_key.to_s
|
|
30
|
-
|
|
31
|
-
@
|
|
32
|
-
|
|
69
|
+
@api_key = api_key.to_s.strip
|
|
70
|
+
raise ArgumentError, "Evaluation API key is required" if @api_key.empty?
|
|
71
|
+
unless @api_key.match?(/\A[\x21-\x7E]+\z/)
|
|
72
|
+
raise ArgumentError, "Evaluation API key must contain only visible ASCII characters"
|
|
73
|
+
end
|
|
74
|
+
|
|
75
|
+
@timeout = Float(timeout, exception: false)
|
|
76
|
+
@open_timeout = Float(open_timeout, exception: false)
|
|
77
|
+
unless [ @timeout, @open_timeout ].all? { |value| value&.finite? && value.positive? }
|
|
33
78
|
raise ArgumentError, "Evaluation delivery timeouts must be positive and finite"
|
|
34
79
|
end
|
|
35
80
|
rescue URI::InvalidURIError
|
|
@@ -43,7 +88,7 @@ module ActiveAgent
|
|
|
43
88
|
identities.each do |key, value|
|
|
44
89
|
raise ArgumentError, "#{key} must be a nonempty string" unless value.is_a?(String) && !value.strip.empty?
|
|
45
90
|
end
|
|
46
|
-
body =
|
|
91
|
+
body = encode(identities.merge("version" => 1, "report" => report_hash(report)))
|
|
47
92
|
raise Error, "Evaluation report exceeds the 2 MiB delivery limit; publish a smaller selection" if body.bytesize > MAX_BYTES
|
|
48
93
|
|
|
49
94
|
http = Net::HTTP.new(@uri.hostname, @uri.port)
|
|
@@ -57,19 +102,70 @@ module ActiveAgent
|
|
|
57
102
|
request["Accept"] = "application/json"
|
|
58
103
|
request.body = body
|
|
59
104
|
response = http.request(request)
|
|
60
|
-
unless %w[200 201].include?(response.code)
|
|
61
|
-
raise Error, "Evaluation delivery rejected (HTTP #{response.code}); retain the report and run_id for retry"
|
|
62
|
-
end
|
|
105
|
+
raise rejection(response) unless %w[200 201].include?(response.code)
|
|
63
106
|
|
|
64
|
-
receipt = JSON.parse(response.body)
|
|
107
|
+
receipt = JSON.parse(response.body.to_s)
|
|
65
108
|
unless receipt.is_a?(Hash) && receipt["run_id"] == run_id && receipt["status"] == "complete" && receipt["id"] && receipt["evaluation_id"]
|
|
66
|
-
raise Error
|
|
109
|
+
raise Error.new("Evaluation collector returned an invalid completion receipt; retain the report and run_id for retry", retryable: true)
|
|
67
110
|
end
|
|
68
111
|
receipt
|
|
69
112
|
rescue JSON::ParserError
|
|
70
|
-
|
|
113
|
+
# The parser's message quotes the body.
|
|
114
|
+
raise Error.new("Evaluation collector returned invalid JSON; retain the report and run_id for retry", retryable: true), cause: nil
|
|
115
|
+
rescue Net::HTTPBadResponse, Net::HTTPHeaderSyntaxError, Zlib::Error => e
|
|
116
|
+
# These messages can quote the response's status line, headers or body.
|
|
117
|
+
raise Error.new("Evaluation collector returned a malformed response (#{e.class}); retain the report and run_id for retry", retryable: true), cause: nil
|
|
71
118
|
rescue IOError, SocketError, SystemCallError, Timeout::Error, OpenSSL::SSL::SSLError => e
|
|
72
|
-
raise Error
|
|
119
|
+
raise Error.new("Evaluation delivery failed (#{e.class}); retain the report and run_id for retry", retryable: true)
|
|
120
|
+
end
|
|
121
|
+
|
|
122
|
+
private
|
|
123
|
+
|
|
124
|
+
def report_hash(report)
|
|
125
|
+
hash = report.to_h if report.respond_to?(:to_h) && !report.nil? && !report.is_a?(Array)
|
|
126
|
+
raise ArgumentError, "report must be a Report or its saved JSON hash" unless hash.is_a?(Hash)
|
|
127
|
+
|
|
128
|
+
hash
|
|
129
|
+
end
|
|
130
|
+
|
|
131
|
+
# Returns the envelope as JSON. A report holding invalid UTF-8, NaN,
|
|
132
|
+
# Infinity or nesting over JSON's depth limit raises a non-retryable
|
|
133
|
+
# Error, without the generator's message, which can quote the report.
|
|
134
|
+
def encode(envelope)
|
|
135
|
+
JSON.generate(envelope)
|
|
136
|
+
rescue JSON::JSONError, EncodingError => e
|
|
137
|
+
raise Error.new("Evaluation report cannot be encoded as JSON (#{e.class}); correct the report before publishing"), cause: nil
|
|
138
|
+
end
|
|
139
|
+
|
|
140
|
+
def rejection(response)
|
|
141
|
+
status = response.code.to_i
|
|
142
|
+
retryable, guidance = REJECTIONS.fetch(status) do
|
|
143
|
+
if status == 408 || (500..599).cover?(status)
|
|
144
|
+
[ true, "retain the report and run_id for retry" ]
|
|
145
|
+
else
|
|
146
|
+
[ false, "retain the report and run_id, and resolve the rejection before retrying" ]
|
|
147
|
+
end
|
|
148
|
+
end
|
|
149
|
+
detail = collector_detail(response.body)
|
|
150
|
+
reason = detail ? "HTTP #{status}: #{detail}" : "HTTP #{status}"
|
|
151
|
+
Error.new("Evaluation delivery rejected (#{reason}); #{guidance}", status: status, detail: detail, retryable: retryable)
|
|
152
|
+
end
|
|
153
|
+
|
|
154
|
+
# Returns the +error+ string of a JSON object body, or nil for any other
|
|
155
|
+
# body. Control characters and runs of whitespace become one space, the
|
|
156
|
+
# API key becomes [FILTERED], and the result is cut to DETAIL_LIMIT
|
|
157
|
+
# characters: `"answer\n\tis required"` → `"answer is required"`.
|
|
158
|
+
def collector_detail(body)
|
|
159
|
+
parsed = JSON.parse(body.to_s)
|
|
160
|
+
message = parsed["error"] if parsed.is_a?(Hash)
|
|
161
|
+
return unless message.is_a?(String)
|
|
162
|
+
|
|
163
|
+
detail = message.scrub.gsub(/[[:space:]\p{C}]+/, " ").strip.gsub(@api_key, "[FILTERED]")
|
|
164
|
+
return if detail.empty?
|
|
165
|
+
|
|
166
|
+
detail.length > DETAIL_LIMIT ? "#{detail[0, DETAIL_LIMIT - 1].rstrip}…" : detail
|
|
167
|
+
rescue JSON::ParserError, EncodingError
|
|
168
|
+
nil
|
|
73
169
|
end
|
|
74
170
|
end
|
|
75
171
|
end
|
|
@@ -16,10 +16,10 @@ module ActiveAgent
|
|
|
16
16
|
# faults in `refine_faults` (up to `judge_limit` calls per run).
|
|
17
17
|
#
|
|
18
18
|
# Runner.new(
|
|
19
|
-
# scenarios: suite.scenarios(groups: %w[
|
|
19
|
+
# scenarios: suite.scenarios(groups: %w[history]),
|
|
20
20
|
# models: ModelSpec.parse_all(%w[gpt-5-mini claude-haiku-4-5], default_provider: "openai"),
|
|
21
21
|
# criteria: [{ "key" => "response_present", "type" => "response_present" }],
|
|
22
|
-
# available_tools: { "
|
|
22
|
+
# available_tools: { "lookup_order" => "Find an order by number" },
|
|
23
23
|
# instructions: agent.instructions,
|
|
24
24
|
# judge: Judge.new(label: "claude-opus-5") { |instructions:, prompt:| ... },
|
|
25
25
|
# replay: ->(scenario, spec) { ... },
|
|
@@ -7,11 +7,11 @@ module ActiveAgent
|
|
|
7
7
|
# answer is expected to do.
|
|
8
8
|
#
|
|
9
9
|
# @!attribute key
|
|
10
|
-
# @return [String] stable within a suite ("
|
|
10
|
+
# @return [String] stable within a suite ("history_3"), so results line up across runs
|
|
11
11
|
# @!attribute group
|
|
12
|
-
# @return [String, nil] the group key ("
|
|
12
|
+
# @return [String, nil] the group key ("history")
|
|
13
13
|
# @!attribute group_name
|
|
14
|
-
# @return [String, nil] the group's display name ("
|
|
14
|
+
# @return [String, nil] the group's display name ("History / audit")
|
|
15
15
|
# @!attribute expected_tools
|
|
16
16
|
# @return [Array<String>] tool names a passing answer calls (any one of them)
|
|
17
17
|
# @!attribute expected_patterns
|
|
@@ -9,8 +9,8 @@ module ActiveAgent
|
|
|
9
9
|
# - `# Heading` or `**Heading**` lines, which start a group; so does a
|
|
10
10
|
# short unmarked line ending in a colon (`Find records:`)
|
|
11
11
|
# - a message in backticks at the start of the line, followed by notes,
|
|
12
|
-
# as in an issue
|
|
13
|
-
# `` 3. `Show me all
|
|
12
|
+
# as in a list copied from an issue:
|
|
13
|
+
# `` 3. `Show me all tickets with no assignee` — 12 in the sample data ``
|
|
14
14
|
# - trailing ` | tools: a, b | contains: x | not_contains: y | key: k`
|
|
15
15
|
# options on a line
|
|
16
16
|
# - a JSON array of strings, or of objects with `prompt` (or `message`),
|
|
@@ -19,7 +19,7 @@ module ActiveAgent
|
|
|
19
19
|
# group names, stable keys, notes and production-only flags
|
|
20
20
|
#
|
|
21
21
|
# Every scenario gets a key unique within the paste, derived from its group
|
|
22
|
-
# and position ("
|
|
22
|
+
# and position ("history_3"), unless the line names one. The result is an
|
|
23
23
|
# array of string-keyed hashes; `Scenario.from_hash` builds the structs.
|
|
24
24
|
class ScenarioParser
|
|
25
25
|
class ParseError < ArgumentError; end
|
|
@@ -8,17 +8,17 @@ module ActiveAgent
|
|
|
8
8
|
# suite. That is how an app keeps a shared suite and lets a deployment add or
|
|
9
9
|
# reword questions.
|
|
10
10
|
#
|
|
11
|
-
# suite:
|
|
12
|
-
# description:
|
|
11
|
+
# suite: support_desk
|
|
12
|
+
# description: Questions the support team asks every week
|
|
13
13
|
# groups:
|
|
14
|
-
# - key:
|
|
15
|
-
# name:
|
|
14
|
+
# - key: open_tickets
|
|
15
|
+
# name: Open tickets
|
|
16
16
|
# scenarios:
|
|
17
|
-
# - key:
|
|
18
|
-
# prompt: Which
|
|
17
|
+
# - key: open_tickets_1
|
|
18
|
+
# prompt: Which open tickets mention a refund?
|
|
19
19
|
# expect:
|
|
20
|
-
# tools: [
|
|
21
|
-
# notes:
|
|
20
|
+
# tools: [find_tickets, count_tickets]
|
|
21
|
+
# notes: The sample data has three.
|
|
22
22
|
# production_only: false
|
|
23
23
|
class Suite
|
|
24
24
|
class NotFound < StandardError; end
|
data/lib/active_agent/evals.rb
CHANGED
|
@@ -50,7 +50,7 @@ require_relative "evals/publisher"
|
|
|
50
50
|
# report = ActiveAgent::Evals::Runner.new(
|
|
51
51
|
# scenarios: scenarios,
|
|
52
52
|
# models: models,
|
|
53
|
-
# available_tools: { "
|
|
53
|
+
# available_tools: { "lookup_order" => "Find an order by number" },
|
|
54
54
|
# replay: ->(scenario, spec) { my_agent.run(scenario.prompt, model: spec.model, provider: spec.provider) }
|
|
55
55
|
# ).call
|
|
56
56
|
#
|
|
@@ -23,7 +23,21 @@ module ActiveAgent
|
|
|
23
23
|
# argument, so a record arrives as the same record the caller passed and
|
|
24
24
|
# the agent's authorization callbacks decide against a real user rather
|
|
25
25
|
# than against nil.
|
|
26
|
-
|
|
26
|
+
#
|
|
27
|
+
# +direct_generation_type+ marks a job enqueued by Agent.prompt(...) or
|
|
28
|
+
# Agent.embed(...), which have no action to call: +agent_method+ is then
|
|
29
|
+
# the synthetic +__direct_*__+ name, so the generation is rebuilt from
|
|
30
|
+
# +direct_args+ and +direct_options+ instead.
|
|
31
|
+
def perform(agent, agent_method, generation_method, args:, kwargs: nil, params: nil, actor: nil,
|
|
32
|
+
direct_generation_type: nil, direct_args: nil, direct_options: nil)
|
|
33
|
+
if direct_generation_type
|
|
34
|
+
generation = ActiveAgent::Parameterized::DirectGeneration.new(
|
|
35
|
+
agent.constantize, direct_generation_type.to_sym, params || {},
|
|
36
|
+
*Array(direct_args), **(direct_options || {}).symbolize_keys
|
|
37
|
+
)
|
|
38
|
+
return generation.public_send(generation_method)
|
|
39
|
+
end
|
|
40
|
+
|
|
27
41
|
agent_class = params ? agent.constantize.with(params) : agent.constantize
|
|
28
42
|
agent_class = agent_class.as(actor) if actor
|
|
29
43
|
prompt = if kwargs
|
|
@@ -10,7 +10,8 @@ require_relative "concerns/tool_choice_clearing"
|
|
|
10
10
|
GEM_LOADERS = {
|
|
11
11
|
anthropic: [ "anthropic", "~> 1.12", "anthropic" ],
|
|
12
12
|
openai: [ "openai", "~> 0.34", "openai" ],
|
|
13
|
-
|
|
13
|
+
# ruby_llm 2.0 renamed the APIs RubyLLMProvider calls.
|
|
14
|
+
ruby_llm: [ "ruby_llm", "~> 1.0", "ruby_llm" ]
|
|
14
15
|
}
|
|
15
16
|
|
|
16
17
|
# Requires a provider's gem dependency.
|
|
@@ -18,7 +19,8 @@ GEM_LOADERS = {
|
|
|
18
19
|
# @param type [Symbol] provider type (:anthropic, :openai)
|
|
19
20
|
# @param file_name [String] for error context
|
|
20
21
|
# @return [void]
|
|
21
|
-
# @raise [LoadError] when
|
|
22
|
+
# @raise [LoadError] when the gem is not installed, or when the loaded
|
|
23
|
+
# version is outside the supported range
|
|
22
24
|
def require_gem!(type, file_name)
|
|
23
25
|
gem_name, requirement, package_name = GEM_LOADERS.fetch(type)
|
|
24
26
|
provider_name = file_name.split("/").last.delete_suffix(".rb").camelize
|
|
@@ -27,6 +29,12 @@ def require_gem!(type, file_name)
|
|
|
27
29
|
gem(gem_name, requirement)
|
|
28
30
|
require(package_name)
|
|
29
31
|
rescue LoadError
|
|
32
|
+
loaded = Gem.loaded_specs[gem_name]
|
|
33
|
+
if loaded && !Gem::Requirement.new(requirement).satisfied_by?(loaded.version)
|
|
34
|
+
raise LoadError, "#{provider_name} supports the '#{gem_name}' gem #{requirement}, but #{loaded.version} is loaded. " \
|
|
35
|
+
"Add `gem \"#{gem_name}\", \"#{requirement}\"` to your Gemfile and run `bundle update #{gem_name}`."
|
|
36
|
+
end
|
|
37
|
+
|
|
30
38
|
raise LoadError, "The '#{gem_name}' gem is required for #{provider_name}. Please add it to your Gemfile and run `bundle install`."
|
|
31
39
|
end
|
|
32
40
|
end
|
|
@@ -38,10 +38,20 @@ module ActiveAgent
|
|
|
38
38
|
end
|
|
39
39
|
|
|
40
40
|
# Handle mcps parameter (common format) -> transforms to mcp_servers (provider format)
|
|
41
|
-
if params[:mcps]
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
41
|
+
if params[:mcps] || params[:mcp_servers]
|
|
42
|
+
mcps = if params[:mcps]
|
|
43
|
+
params.delete(:mcps)
|
|
44
|
+
else
|
|
45
|
+
params[:mcp_servers]
|
|
46
|
+
end
|
|
47
|
+
|
|
48
|
+
params[:mcp_servers] = normalize_mcp_servers(mcps)
|
|
49
|
+
|
|
50
|
+
# A server's allowed_tools become an mcp_toolset entry beside the
|
|
51
|
+
# request's own tools. Added only when there is one: an empty
|
|
52
|
+
# list would send `"tools": null` on every MCP request.
|
|
53
|
+
mcp_tools = normalize_mcp_tools(mcps)
|
|
54
|
+
params[:tools] = Array(params[:tools]) + mcp_tools if mcp_tools.present?
|
|
45
55
|
end
|
|
46
56
|
|
|
47
57
|
params
|
|
@@ -111,6 +121,49 @@ module ActiveAgent
|
|
|
111
121
|
end
|
|
112
122
|
end
|
|
113
123
|
|
|
124
|
+
# Builds Anthropic mcp_toolset tool entries from MCP server allowed_tools.
|
|
125
|
+
#
|
|
126
|
+
# Accepts allowed_tools entries as tool-name strings/symbols or hashes
|
|
127
|
+
# with a :name key; entries in any other format are ignored.
|
|
128
|
+
#
|
|
129
|
+
# @param mcp_servers [Array<Hash>]
|
|
130
|
+
# @return [Array<Hash>, nil] toolset entries, or nil when none were extracted
|
|
131
|
+
def normalize_mcp_tools(mcp_servers)
|
|
132
|
+
return nil unless mcp_servers.is_a?(Array)
|
|
133
|
+
|
|
134
|
+
result = mcp_servers.filter_map do |server|
|
|
135
|
+
next unless server.is_a?(Hash)
|
|
136
|
+
|
|
137
|
+
server_hash = server.deep_symbolize_keys
|
|
138
|
+
allowed_tools = server_hash[:allowed_tools]
|
|
139
|
+
next unless allowed_tools.is_a?(Array)
|
|
140
|
+
|
|
141
|
+
configs = allowed_tools.filter_map { |tool|
|
|
142
|
+
name = case tool
|
|
143
|
+
when String, Symbol
|
|
144
|
+
tool.to_s
|
|
145
|
+
when Hash
|
|
146
|
+
(tool[:name] || tool["name"]).to_s
|
|
147
|
+
end
|
|
148
|
+
|
|
149
|
+
[ name, { enabled: true } ] if name.present?
|
|
150
|
+
}.to_h
|
|
151
|
+
|
|
152
|
+
next if configs.empty?
|
|
153
|
+
|
|
154
|
+
{
|
|
155
|
+
type: "mcp_toolset",
|
|
156
|
+
mcp_server_name: server_hash[:name],
|
|
157
|
+
default_config: {
|
|
158
|
+
enabled: false
|
|
159
|
+
},
|
|
160
|
+
configs: configs
|
|
161
|
+
}
|
|
162
|
+
end
|
|
163
|
+
|
|
164
|
+
result.presence
|
|
165
|
+
end
|
|
166
|
+
|
|
114
167
|
# Normalizes tool_choice from common format to Anthropic gem model objects.
|
|
115
168
|
#
|
|
116
169
|
# The Anthropic gem expects tool_choice to be a model object (ToolChoiceAuto,
|
|
@@ -96,6 +96,14 @@ module ActiveAgent
|
|
|
96
96
|
server_url: server_hash[:url] || server_hash[:server_url]
|
|
97
97
|
}
|
|
98
98
|
|
|
99
|
+
if server_hash[:require_approval]
|
|
100
|
+
result[:require_approval] = server_hash[:require_approval]
|
|
101
|
+
end
|
|
102
|
+
|
|
103
|
+
if server_hash[:allowed_tools]
|
|
104
|
+
result[:allowed_tools] = server_hash[:allowed_tools]
|
|
105
|
+
end
|
|
106
|
+
|
|
99
107
|
# Keep authorization field (OpenAI uses 'authorization', not 'authorization_token')
|
|
100
108
|
if server_hash[:authorization]
|
|
101
109
|
result[:authorization] = server_hash[:authorization]
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
require_relative "_base_provider"
|
|
4
4
|
|
|
5
|
-
require_gem!(:ruby_llm, __FILE__)
|
|
5
|
+
require_gem!(:ruby_llm, __FILE__)
|
|
6
6
|
|
|
7
7
|
require_relative "ruby_llm/_types"
|
|
8
8
|
require_relative "ruby_llm/tool_proxy"
|
|
@@ -59,7 +59,8 @@ module ActiveAgent
|
|
|
59
59
|
tools: tools || {},
|
|
60
60
|
temperature: parameters[:temperature]
|
|
61
61
|
}
|
|
62
|
-
|
|
62
|
+
schema = ruby_llm_schema(parameters[:response_format])
|
|
63
|
+
kwargs[:schema] = schema if schema
|
|
63
64
|
|
|
64
65
|
# Pass extra params (max_tokens, etc.) via RubyLLM's params: deep-merge
|
|
65
66
|
max_tokens = parameters[:max_tokens] || options.max_tokens
|
|
@@ -345,12 +346,61 @@ module ActiveAgent
|
|
|
345
346
|
call = ::RubyLLM::ToolCall.new(
|
|
346
347
|
id: id,
|
|
347
348
|
name: tc.dig(:function, :name) || tc[:name],
|
|
348
|
-
arguments: tc.dig(:function, :arguments) || tc[:input]
|
|
349
|
+
arguments: ruby_llm_tool_arguments(tc.dig(:function, :arguments) || tc[:input])
|
|
349
350
|
)
|
|
350
351
|
hash[id] = call
|
|
351
352
|
end
|
|
352
353
|
end
|
|
353
354
|
|
|
355
|
+
# ActiveAgent keeps a tool call's arguments as the JSON string the
|
|
356
|
+
# model sent; RubyLLM takes them as a Hash and renders them itself
|
|
357
|
+
# (JSON-encoded for OpenAI, as the input object for Anthropic).
|
|
358
|
+
#
|
|
359
|
+
# A string that is not valid JSON, such as arguments a model cut off
|
|
360
|
+
# mid-way in a stored conversation, is passed through unchanged.
|
|
361
|
+
#
|
|
362
|
+
# @param arguments [String, Hash, nil]
|
|
363
|
+
# @return [Hash, String]
|
|
364
|
+
def ruby_llm_tool_arguments(arguments)
|
|
365
|
+
case arguments
|
|
366
|
+
when Hash
|
|
367
|
+
arguments.deep_stringify_keys
|
|
368
|
+
when String
|
|
369
|
+
arguments.blank? ? {} : JSON.parse(arguments)
|
|
370
|
+
else
|
|
371
|
+
{}
|
|
372
|
+
end
|
|
373
|
+
rescue JSON::ParserError
|
|
374
|
+
arguments
|
|
375
|
+
end
|
|
376
|
+
|
|
377
|
+
# Converts ActiveAgent's response_format to the schema RubyLLM's
|
|
378
|
+
# complete takes, { name:, schema:, strict: }, filling the name and
|
|
379
|
+
# strict flag in the way RubyLLM::Chat#with_schema does.
|
|
380
|
+
#
|
|
381
|
+
# @param response_format [Hash, Symbol, String, nil] ActiveAgent common format
|
|
382
|
+
# @return [Hash, nil] nil when the response is plain text
|
|
383
|
+
# @raise [ArgumentError] for json_schema without a schema, and for any
|
|
384
|
+
# other type, including json_object, which RubyLLM has no mode for
|
|
385
|
+
def ruby_llm_schema(response_format)
|
|
386
|
+
return nil if response_format.nil?
|
|
387
|
+
|
|
388
|
+
format = response_format.is_a?(Hash) ? response_format : { type: response_format.to_s }
|
|
389
|
+
|
|
390
|
+
case format[:type].to_s
|
|
391
|
+
when "text"
|
|
392
|
+
nil
|
|
393
|
+
when "json_schema"
|
|
394
|
+
json_schema = format[:json_schema] || {}
|
|
395
|
+
raise ArgumentError, "RubyLLMProvider needs a schema for a json_schema response_format" unless json_schema[:schema]
|
|
396
|
+
|
|
397
|
+
{ name: json_schema[:name] || "response", schema: json_schema[:schema], strict: json_schema[:strict] != false }
|
|
398
|
+
else
|
|
399
|
+
raise ArgumentError, "RubyLLMProvider supports a json_schema or text response_format, not #{format[:type].inspect}; " \
|
|
400
|
+
"ruby_llm has no JSON object mode, so give json_schema a schema instead"
|
|
401
|
+
end
|
|
402
|
+
end
|
|
403
|
+
|
|
354
404
|
# Converts ActiveAgent tool definitions to RubyLLM ToolProxy objects.
|
|
355
405
|
#
|
|
356
406
|
# @param tools [Array<Hash>, nil] ActiveAgent tool definitions
|
data/lib/active_agent/version.rb
CHANGED
metadata
CHANGED
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
--- !ruby/object:Gem::Specification
|
|
2
2
|
name: activeagent
|
|
3
3
|
version: !ruby/object:Gem::Version
|
|
4
|
-
version: 1.
|
|
4
|
+
version: 1.7.1
|
|
5
5
|
platform: ruby
|
|
6
6
|
authors:
|
|
7
7
|
- Justin Bowen
|
|
8
8
|
autorequire:
|
|
9
9
|
bindir: bin
|
|
10
10
|
cert_chain: []
|
|
11
|
-
date: 2026-09-
|
|
11
|
+
date: 2026-09-30 00:00:00.000000000 Z
|
|
12
12
|
dependencies:
|
|
13
13
|
- !ruby/object:Gem::Dependency
|
|
14
14
|
name: actionpack
|
|
@@ -204,14 +204,14 @@ dependencies:
|
|
|
204
204
|
name: ruby_llm
|
|
205
205
|
requirement: !ruby/object:Gem::Requirement
|
|
206
206
|
requirements:
|
|
207
|
-
- - "
|
|
207
|
+
- - "~>"
|
|
208
208
|
- !ruby/object:Gem::Version
|
|
209
209
|
version: '1.0'
|
|
210
210
|
type: :development
|
|
211
211
|
prerelease: false
|
|
212
212
|
version_requirements: !ruby/object:Gem::Requirement
|
|
213
213
|
requirements:
|
|
214
|
-
- - "
|
|
214
|
+
- - "~>"
|
|
215
215
|
- !ruby/object:Gem::Version
|
|
216
216
|
version: '1.0'
|
|
217
217
|
- !ruby/object:Gem::Dependency
|