actionagent 1.7.0 → 1.7.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/app/assets/builds/action_agent.js +55 -55
- data/app/controllers/action_agent/api/agents_controller.rb +20 -1
- data/app/controllers/action_agent/api/evaluations_controller.rb +36 -1
- data/app/controllers/action_agent/api/provider_models_controller.rb +35 -4
- data/app/services/action_agent/agent_execution_service.rb +18 -2
- data/app/services/action_agent/evaluation_runner_service.rb +22 -9
- data/app/services/action_agent/evaluation_tool_resolver.rb +3 -3
- data/app/services/action_agent/scenario_evaluation_runner.rb +8 -4
- data/config/routes.rb +3 -0
- data/lib/action_agent/version.rb +1 -1
- metadata +2 -2
|
@@ -18,6 +18,8 @@ module ActionAgent
|
|
|
18
18
|
DEFAULT_LIST_SORT = "recent"
|
|
19
19
|
# Conversations returned to the runner's picker when no limit is asked for.
|
|
20
20
|
CONVERSATIONS_LIMIT = 50
|
|
21
|
+
# Model names returned by #recorded_models.
|
|
22
|
+
RECORDED_MODELS_LIMIT = 50
|
|
21
23
|
# Keywords Agent#execute takes in its own right, which per-run overrides
|
|
22
24
|
# must never supply (see #execution_params). `actor` is here for the
|
|
23
25
|
# same reason as the rest and one more: a keyword splat wins over the
|
|
@@ -27,7 +29,7 @@ module ActionAgent
|
|
|
27
29
|
|
|
28
30
|
before_action :set_agent, only: [
|
|
29
31
|
:show, :update, :destroy, :versions, :runs, :execute, :test, :restore, :duplicate, :export, :analytics,
|
|
30
|
-
:tool_roster, :conversations, :create_conversation
|
|
32
|
+
:tool_roster, :conversations, :create_conversation, :recorded_models
|
|
31
33
|
]
|
|
32
34
|
before_action :require_execution_enabled!, only: [ :execute, :test ]
|
|
33
35
|
before_action :require_owner!, only: [ :execute, :test ]
|
|
@@ -274,6 +276,23 @@ module ActionAgent
|
|
|
274
276
|
).as_json
|
|
275
277
|
end
|
|
276
278
|
|
|
279
|
+
# GET /api/agents/:id/recorded_models
|
|
280
|
+
#
|
|
281
|
+
# The model names this agent's generations were recorded under, most
|
|
282
|
+
# recently used first. An evaluation without scenarios compares the
|
|
283
|
+
# generations recorded under each name it is given, so these are the
|
|
284
|
+
# names its models field suggests.
|
|
285
|
+
def recorded_models
|
|
286
|
+
generations = AgentGeneration.arel_table
|
|
287
|
+
models = @agent.generations.where.not(model: [ nil, "" ])
|
|
288
|
+
.group(generations[:model])
|
|
289
|
+
.order(Arel::Nodes::Descending.new(generations[:created_at].maximum))
|
|
290
|
+
.limit(RECORDED_MODELS_LIMIT)
|
|
291
|
+
.pluck(generations[:model])
|
|
292
|
+
|
|
293
|
+
render json: { models: models }
|
|
294
|
+
end
|
|
295
|
+
|
|
277
296
|
# GET /api/agents/:id/analytics
|
|
278
297
|
#
|
|
279
298
|
# Every execution of this agent, whoever ran it — the same merged model
|
|
@@ -38,12 +38,28 @@ module ActionAgent
|
|
|
38
38
|
# 50 client-side hides an agent whose evaluations are not among the account's
|
|
39
39
|
# 50 most recent. The scope is already restricted to the current user's
|
|
40
40
|
# agents, so an id outside it simply returns nothing.
|
|
41
|
+
#
|
|
42
|
+
# Three fields feed the dashboard's model pickers. They describe the
|
|
43
|
+
# credentials of #picker_credentials_owner:
|
|
44
|
+
# - judge_provider: the provider a judge model runs on, null
|
|
45
|
+
# when none has credentials or the lookup
|
|
46
|
+
# raised
|
|
47
|
+
# - judge_provider_error: true when that lookup raised, as it does
|
|
48
|
+
# when a stored key no longer decrypts
|
|
49
|
+
# - model_providers: the providers agent runs have credentials
|
|
50
|
+
# for, leaving out any whose credentials
|
|
51
|
+
# cannot be read
|
|
41
52
|
def index
|
|
42
53
|
scope = evaluations_scope
|
|
43
54
|
scope = scope.where(agent_id: params[:agent_id]) if params[:agent_id].present?
|
|
44
55
|
evaluations = scope.includes(:agent, :evaluation_runs, :scenarios).recent.limit(50)
|
|
56
|
+
owner = picker_credentials_owner
|
|
45
57
|
|
|
46
|
-
render json: {
|
|
58
|
+
render json: {
|
|
59
|
+
evaluations: evaluations.map { |evaluation| serialize(evaluation) },
|
|
60
|
+
**judge_provider_fields(owner),
|
|
61
|
+
model_providers: AgentExecutionService.available_providers(owner)
|
|
62
|
+
}
|
|
47
63
|
end
|
|
48
64
|
|
|
49
65
|
# Runs listed per evaluation on GET /api/evaluations/:id. The rest of
|
|
@@ -224,6 +240,25 @@ module ActionAgent
|
|
|
224
240
|
evaluation.evaluation_runs.create!(status: :failed, error_message: e.message, completed_at: Time.current)
|
|
225
241
|
end
|
|
226
242
|
|
|
243
|
+
# Returns whose credentials the index's model picker fields describe.
|
|
244
|
+
# Agent runs and their judge use the evaluated agent's owner's
|
|
245
|
+
# credentials, so a list scoped to one agent reads that agent's owner,
|
|
246
|
+
# and an unscoped list the signed-in owner.
|
|
247
|
+
def picker_credentials_owner
|
|
248
|
+
agent = owner_agents.find_by(id: params[:agent_id]) if params[:agent_id].present?
|
|
249
|
+
agent ? agent.owner : current_owner
|
|
250
|
+
end
|
|
251
|
+
|
|
252
|
+
# Returns the index's judge_provider and judge_provider_error for
|
|
253
|
+
# +owner+. A lookup that raises, from a key that no longer decrypts or a
|
|
254
|
+
# host credentials hook that fails, is logged and reported as an error.
|
|
255
|
+
def judge_provider_fields(owner)
|
|
256
|
+
{ judge_provider: EvaluationRunnerService.judge_provider_for(owner)&.to_s, judge_provider_error: false }
|
|
257
|
+
rescue StandardError => e
|
|
258
|
+
Rails.logger.warn("[Evaluations] judge provider lookup failed: #{e.class}: #{e.message}")
|
|
259
|
+
{ judge_provider: nil, judge_provider_error: true }
|
|
260
|
+
end
|
|
261
|
+
|
|
227
262
|
def evaluations_scope
|
|
228
263
|
Evaluation.joins(:agent).where(agent: owner_agents)
|
|
229
264
|
end
|
|
@@ -2,14 +2,22 @@
|
|
|
2
2
|
|
|
3
3
|
module ActionAgent
|
|
4
4
|
module Api
|
|
5
|
-
# Model catalogs for the agent builder
|
|
5
|
+
# Model catalogs for the dashboard's model pickers: the agent builder and
|
|
6
|
+
# editor, and the evaluation forms.
|
|
6
7
|
#
|
|
7
8
|
# Hosted providers get a curated list of current models (kept here, server
|
|
8
9
|
# side, so the UI can't drift stale). Ollama is queried live from the
|
|
9
10
|
# account's configured host (Settings -> Provider API Keys, falling back to
|
|
10
11
|
# the platform config) so locally pulled models appear; OpenRouter is
|
|
11
|
-
# queried from its public catalog
|
|
12
|
-
# list on any
|
|
12
|
+
# queried from its public catalog, and Anthropic from its Models API with
|
|
13
|
+
# the account's key. Live lookups fall back to the curated list on any
|
|
14
|
+
# failure.
|
|
15
|
+
#
|
|
16
|
+
# When the host app loads RubyLLM, the chat models its registry lists for
|
|
17
|
+
# the provider that take and return text follow: RubyLLM's bundled
|
|
18
|
+
# catalog, or the host's own model table when it configured one. The list
|
|
19
|
+
# stays de-duplicated, and its first id, the builder's preselected
|
|
20
|
+
# default, is always the live or curated one.
|
|
13
21
|
class ProviderModelsController < BaseController
|
|
14
22
|
before_action :require_owner!
|
|
15
23
|
|
|
@@ -42,7 +50,7 @@ module ActionAgent
|
|
|
42
50
|
source = "curated"
|
|
43
51
|
end
|
|
44
52
|
|
|
45
|
-
render json: { provider: provider, models: models, source: source }
|
|
53
|
+
render json: { provider: provider, models: (models + registry_models(provider)).uniq, source: source }
|
|
46
54
|
end
|
|
47
55
|
|
|
48
56
|
private
|
|
@@ -106,6 +114,29 @@ module ActionAgent
|
|
|
106
114
|
nil
|
|
107
115
|
end
|
|
108
116
|
|
|
117
|
+
# Returns the ids of the chat models RubyLLM's registry lists for
|
|
118
|
+
# +provider+ that take and return text. Empty without RubyLLM, or when
|
|
119
|
+
# the registry raises. This engine's provider names are RubyLLM's own
|
|
120
|
+
# slugs.
|
|
121
|
+
#
|
|
122
|
+
# `chat_models` alone is not enough: RubyLLM counts a model as a chat
|
|
123
|
+
# model whenever its output modalities name no other kind, none at all
|
|
124
|
+
# included. That takes in the speech, transcription, moderation and
|
|
125
|
+
# completion-only models its registry lists with no modalities, and the
|
|
126
|
+
# transcription models it lists as taking audio. The few chat models it
|
|
127
|
+
# lists with no modalities are left out with them.
|
|
128
|
+
def registry_models(provider)
|
|
129
|
+
return [] unless defined?(::RubyLLM) && ::RubyLLM.respond_to?(:models)
|
|
130
|
+
|
|
131
|
+
::RubyLLM.models.by_provider(provider).chat_models.filter_map do |model|
|
|
132
|
+
modalities = model.modalities
|
|
133
|
+
model.id.presence if Array(modalities&.input).include?("text") && Array(modalities&.output).include?("text")
|
|
134
|
+
end
|
|
135
|
+
rescue StandardError => e
|
|
136
|
+
Rails.logger.warn("[ProviderModels] RubyLLM registry lookup failed: #{e.message}")
|
|
137
|
+
[]
|
|
138
|
+
end
|
|
139
|
+
|
|
109
140
|
def fetch_json(uri)
|
|
110
141
|
response = Net::HTTP.start(
|
|
111
142
|
uri.host, uri.port,
|
|
@@ -49,6 +49,12 @@ module ActionAgent
|
|
|
49
49
|
new(agent_record, run).call
|
|
50
50
|
end
|
|
51
51
|
|
|
52
|
+
# Returns the providers in Agent::PROVIDERS a run on +owner+'s behalf has
|
|
53
|
+
# credentials for, in that order (see #available_providers).
|
|
54
|
+
def self.available_providers(owner)
|
|
55
|
+
new(nil, nil, owner: owner).available_providers
|
|
56
|
+
end
|
|
57
|
+
|
|
52
58
|
# Tool-call keywords that name the caller. The model's arguments and the
|
|
53
59
|
# run's actor share one keyword namespace by the time they reach a tool,
|
|
54
60
|
# so anything a model emits under these names is dropped before the call:
|
|
@@ -56,13 +62,23 @@ module ActionAgent
|
|
|
56
62
|
# documents a model reads are attacker-reachable.
|
|
57
63
|
ACTOR_KEYWORDS = %i[actor current_user].freeze
|
|
58
64
|
|
|
59
|
-
|
|
65
|
+
# +owner+ is whose provider credentials the run uses: the agent record's
|
|
66
|
+
# owner unless given.
|
|
67
|
+
def initialize(agent_record, run, owner: nil)
|
|
60
68
|
@agent_record = agent_record
|
|
61
69
|
@run = run
|
|
70
|
+
@owner = owner
|
|
62
71
|
@tool_invocations = []
|
|
63
72
|
@event_sequence = 0
|
|
64
73
|
end
|
|
65
74
|
|
|
75
|
+
# Returns the providers in Agent::PROVIDERS the owner's credentials, or
|
|
76
|
+
# the host's config, let a run use: #provider_available? for each.
|
|
77
|
+
# @return [Array<String>]
|
|
78
|
+
def available_providers
|
|
79
|
+
Agent::PROVIDERS.select { |name| provider_available?(name) }
|
|
80
|
+
end
|
|
81
|
+
|
|
66
82
|
# The caller this run executes on behalf of, or nil when it runs
|
|
67
83
|
# unattributed. Passed to every tool as +actor:+ — a host's SchemaTools
|
|
68
84
|
# scope block, Pundit policy or agent callback decides what that means.
|
|
@@ -943,7 +959,7 @@ module ActionAgent
|
|
|
943
959
|
# The agent's owner under the configured mode; nil when the install
|
|
944
960
|
# has no owner model at all.
|
|
945
961
|
def owner
|
|
946
|
-
@owner ||= @agent_record
|
|
962
|
+
@owner ||= @agent_record&.owner
|
|
947
963
|
end
|
|
948
964
|
|
|
949
965
|
def record_trace(root_span)
|
|
@@ -15,8 +15,17 @@ module ActionAgent
|
|
|
15
15
|
new(evaluation).call
|
|
16
16
|
end
|
|
17
17
|
|
|
18
|
-
|
|
18
|
+
# Returns the provider +owner+'s evaluation judge runs on (see
|
|
19
|
+
# #judge_provider), or nil when no provider has credentials.
|
|
20
|
+
def self.judge_provider_for(owner)
|
|
21
|
+
new(nil, owner: owner).judge_provider
|
|
22
|
+
end
|
|
23
|
+
|
|
24
|
+
# +owner+ is whose provider credentials the judge uses: the evaluated
|
|
25
|
+
# agent's owner unless given.
|
|
26
|
+
def initialize(evaluation, owner: nil)
|
|
19
27
|
@evaluation = evaluation
|
|
28
|
+
@owner = owner
|
|
20
29
|
end
|
|
21
30
|
|
|
22
31
|
def call
|
|
@@ -71,6 +80,17 @@ module ActionAgent
|
|
|
71
80
|
raise
|
|
72
81
|
end
|
|
73
82
|
|
|
83
|
+
# Returns the provider the judge runs on: the first of Anthropic, OpenAI
|
|
84
|
+
# and OpenRouter the owner or the host's config has a key for, else Ollama
|
|
85
|
+
# when the owner configured a host; nil when none. The judge runs
|
|
86
|
+
# `judge_model` as that provider's own model id.
|
|
87
|
+
def judge_provider
|
|
88
|
+
@judge_provider ||=
|
|
89
|
+
%i[anthropic openai openrouter].find do |name|
|
|
90
|
+
owner_provider_options(name).any? || global_provider_token?(name)
|
|
91
|
+
end || (:ollama if owner_provider_options(:ollama).any?)
|
|
92
|
+
end
|
|
93
|
+
|
|
74
94
|
private
|
|
75
95
|
|
|
76
96
|
# Scores each sample-based criterion once per candidate model cohort and
|
|
@@ -547,13 +567,6 @@ module ActionAgent
|
|
|
547
567
|
judge_provider.present?
|
|
548
568
|
end
|
|
549
569
|
|
|
550
|
-
def judge_provider
|
|
551
|
-
@judge_provider ||=
|
|
552
|
-
%i[anthropic openai openrouter].find do |name|
|
|
553
|
-
owner_provider_options(name).any? || global_provider_token?(name)
|
|
554
|
-
end || (:ollama if owner_provider_options(:ollama).any?)
|
|
555
|
-
end
|
|
556
|
-
|
|
557
570
|
def global_provider_token?(name)
|
|
558
571
|
config = ActiveAgent.configuration[name]
|
|
559
572
|
config.respond_to?(:[]) && config[:access_token].present?
|
|
@@ -571,7 +584,7 @@ module ActionAgent
|
|
|
571
584
|
end
|
|
572
585
|
|
|
573
586
|
def owner
|
|
574
|
-
@owner ||= @evaluation
|
|
587
|
+
@owner ||= @evaluation&.agent&.owner
|
|
575
588
|
end
|
|
576
589
|
|
|
577
590
|
def judge_class
|
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
module ActionAgent
|
|
4
4
|
# Names the MCP server behind a tool an evaluation run needs, for the
|
|
5
5
|
# report's fix items (ActiveAgent::Evals::Report#fix_items): a scenario
|
|
6
|
-
# that expected +
|
|
6
|
+
# that expected +track_shipment+ and never got it is fixed by enabling the
|
|
7
7
|
# server that serves it, and the item can only say so — and deep-link to
|
|
8
8
|
# MCP Services — when something here can name that server.
|
|
9
9
|
#
|
|
@@ -97,7 +97,7 @@ module ActionAgent
|
|
|
97
97
|
end
|
|
98
98
|
|
|
99
99
|
# normalized key => the display name a configured hash entry carries
|
|
100
|
-
# alongside its key ({"key" => "
|
|
100
|
+
# alongside its key ({"key" => "shipping", "name" => "Shipping Desk"}).
|
|
101
101
|
def configured_names
|
|
102
102
|
@configured_names ||= configured_entries.each_with_object({}) do |entry, map|
|
|
103
103
|
next unless entry.respond_to?(:key?)
|
|
@@ -111,7 +111,7 @@ module ActionAgent
|
|
|
111
111
|
end
|
|
112
112
|
|
|
113
113
|
# bare tool name => server key, from configured entries that list the
|
|
114
|
-
# tools they serve ({"name" => "
|
|
114
|
+
# tools they serve ({"name" => "shipping", "tools" => ["track_shipment"]}),
|
|
115
115
|
# in the catalog's own +tool_hints+ spelling or as tool hashes.
|
|
116
116
|
def configured_tools
|
|
117
117
|
@configured_tools ||= configured_entries.each_with_object({}) do |entry, map|
|
|
@@ -21,6 +21,11 @@ module ActionAgent
|
|
|
21
21
|
class ScenarioEvaluationRunner < EvaluationRunnerService
|
|
22
22
|
Evals = ActiveAgent::Evals
|
|
23
23
|
|
|
24
|
+
# The providers a candidate model name resolves against. `mock` is the
|
|
25
|
+
# framework's test double, accepted so the test suite can compare cohorts
|
|
26
|
+
# offline.
|
|
27
|
+
CANDIDATE_PROVIDERS = (Agent::PROVIDERS + %w[mock]).freeze
|
|
28
|
+
|
|
24
29
|
# `run` is an EvaluationRun created ahead of time (by run_later!, so the
|
|
25
30
|
# UI can show it pending while the job waits); absent, one is created here.
|
|
26
31
|
def self.call(evaluation, selection: {}, run: nil)
|
|
@@ -126,15 +131,14 @@ module ActionAgent
|
|
|
126
131
|
end
|
|
127
132
|
|
|
128
133
|
# The models to compare: an explicit selection, else the evaluation's
|
|
129
|
-
# compare_models, else the agent as configured.
|
|
130
|
-
# test double, accepted so the test suite can compare cohorts offline.
|
|
134
|
+
# compare_models, else the agent as configured.
|
|
131
135
|
def model_specs
|
|
132
136
|
names = Array(@selection[:models]).presence || @evaluation.compare_models
|
|
133
|
-
specs = Evals::ModelSpec.parse_all(names, default_provider: @evaluation.agent.provider, providers:
|
|
137
|
+
specs = Evals::ModelSpec.parse_all(names, default_provider: @evaluation.agent.provider, providers: CANDIDATE_PROVIDERS)
|
|
134
138
|
# parse_all resolves a bare name against `providers:` but passes through a
|
|
135
139
|
# `provider/model` whose provider is not in that list, so the run would
|
|
136
140
|
# otherwise reach the replay with a provider nothing can serve.
|
|
137
|
-
unsupported = specs.map(&:provider).uniq -
|
|
141
|
+
unsupported = specs.map(&:provider).uniq - CANDIDATE_PROVIDERS
|
|
138
142
|
raise ArgumentError, "unsupported model provider: #{unsupported.to_sentence}" if unsupported.any?
|
|
139
143
|
|
|
140
144
|
return specs if specs.any?
|
data/config/routes.rb
CHANGED
|
@@ -50,6 +50,9 @@ ActionAgent::Engine.routes.draw do
|
|
|
50
50
|
# and a fresh one to pin a first message to.
|
|
51
51
|
get :conversations
|
|
52
52
|
post :conversations, action: :create_conversation
|
|
53
|
+
# The evaluation form's models field: the model names this agent's
|
|
54
|
+
# generations were recorded under.
|
|
55
|
+
get :recorded_models
|
|
53
56
|
end
|
|
54
57
|
collection do
|
|
55
58
|
get :presets
|
data/lib/action_agent/version.rb
CHANGED
metadata
CHANGED
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
--- !ruby/object:Gem::Specification
|
|
2
2
|
name: actionagent
|
|
3
3
|
version: !ruby/object:Gem::Version
|
|
4
|
-
version: 1.7.
|
|
4
|
+
version: 1.7.1
|
|
5
5
|
platform: ruby
|
|
6
6
|
authors:
|
|
7
7
|
- Justin Bowen
|
|
8
8
|
autorequire:
|
|
9
9
|
bindir: bin
|
|
10
10
|
cert_chain: []
|
|
11
|
-
date: 2026-09-
|
|
11
|
+
date: 2026-09-30 00:00:00.000000000 Z
|
|
12
12
|
dependencies:
|
|
13
13
|
- !ruby/object:Gem::Dependency
|
|
14
14
|
name: activeagent
|