actionagent 1.7.0 → 1.7.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -18,6 +18,8 @@ module ActionAgent
18
18
  DEFAULT_LIST_SORT = "recent"
19
19
  # Conversations returned to the runner's picker when no limit is asked for.
20
20
  CONVERSATIONS_LIMIT = 50
21
+ # Model names returned by #recorded_models.
22
+ RECORDED_MODELS_LIMIT = 50
21
23
  # Keywords Agent#execute takes in its own right, which per-run overrides
22
24
  # must never supply (see #execution_params). `actor` is here for the
23
25
  # same reason as the rest and one more: a keyword splat wins over the
@@ -27,7 +29,7 @@ module ActionAgent
27
29
 
28
30
  before_action :set_agent, only: [
29
31
  :show, :update, :destroy, :versions, :runs, :execute, :test, :restore, :duplicate, :export, :analytics,
30
- :tool_roster, :conversations, :create_conversation
32
+ :tool_roster, :conversations, :create_conversation, :recorded_models
31
33
  ]
32
34
  before_action :require_execution_enabled!, only: [ :execute, :test ]
33
35
  before_action :require_owner!, only: [ :execute, :test ]
@@ -274,6 +276,23 @@ module ActionAgent
274
276
  ).as_json
275
277
  end
276
278
 
279
+ # GET /api/agents/:id/recorded_models
280
+ #
281
+ # The model names this agent's generations were recorded under, most
282
+ # recently used first. An evaluation without scenarios compares the
283
+ # generations recorded under each name it is given, so these are the
284
+ # names its models field suggests.
285
+ def recorded_models
286
+ generations = AgentGeneration.arel_table
287
+ models = @agent.generations.where.not(model: [ nil, "" ])
288
+ .group(generations[:model])
289
+ .order(Arel::Nodes::Descending.new(generations[:created_at].maximum))
290
+ .limit(RECORDED_MODELS_LIMIT)
291
+ .pluck(generations[:model])
292
+
293
+ render json: { models: models }
294
+ end
295
+
277
296
  # GET /api/agents/:id/analytics
278
297
  #
279
298
  # Every execution of this agent, whoever ran it — the same merged model
@@ -38,12 +38,28 @@ module ActionAgent
38
38
  # 50 client-side hides an agent whose evaluations are not among the account's
39
39
  # 50 most recent. The scope is already restricted to the current user's
40
40
  # agents, so an id outside it simply returns nothing.
41
+ #
42
+ # Three fields feed the dashboard's model pickers. They describe the
43
+ # credentials of #picker_credentials_owner:
44
+ # - judge_provider: the provider a judge model runs on, null
45
+ # when none has credentials or the lookup
46
+ # raised
47
+ # - judge_provider_error: true when that lookup raised, as it does
48
+ # when a stored key no longer decrypts
49
+ # - model_providers: the providers agent runs have credentials
50
+ # for, leaving out any whose credentials
51
+ # cannot be read
41
52
  def index
42
53
  scope = evaluations_scope
43
54
  scope = scope.where(agent_id: params[:agent_id]) if params[:agent_id].present?
44
55
  evaluations = scope.includes(:agent, :evaluation_runs, :scenarios).recent.limit(50)
56
+ owner = picker_credentials_owner
45
57
 
46
- render json: { evaluations: evaluations.map { |evaluation| serialize(evaluation) } }
58
+ render json: {
59
+ evaluations: evaluations.map { |evaluation| serialize(evaluation) },
60
+ **judge_provider_fields(owner),
61
+ model_providers: AgentExecutionService.available_providers(owner)
62
+ }
47
63
  end
48
64
 
49
65
  # Runs listed per evaluation on GET /api/evaluations/:id. The rest of
@@ -224,6 +240,25 @@ module ActionAgent
224
240
  evaluation.evaluation_runs.create!(status: :failed, error_message: e.message, completed_at: Time.current)
225
241
  end
226
242
 
243
+ # Returns whose credentials the index's model picker fields describe.
244
+ # Agent runs and their judge use the evaluated agent's owner's
245
+ # credentials, so a list scoped to one agent reads that agent's owner,
246
+ # and an unscoped list the signed-in owner.
247
+ def picker_credentials_owner
248
+ agent = owner_agents.find_by(id: params[:agent_id]) if params[:agent_id].present?
249
+ agent ? agent.owner : current_owner
250
+ end
251
+
252
+ # Returns the index's judge_provider and judge_provider_error for
253
+ # +owner+. A lookup that raises, from a key that no longer decrypts or a
254
+ # host credentials hook that fails, is logged and reported as an error.
255
+ def judge_provider_fields(owner)
256
+ { judge_provider: EvaluationRunnerService.judge_provider_for(owner)&.to_s, judge_provider_error: false }
257
+ rescue StandardError => e
258
+ Rails.logger.warn("[Evaluations] judge provider lookup failed: #{e.class}: #{e.message}")
259
+ { judge_provider: nil, judge_provider_error: true }
260
+ end
261
+
227
262
  def evaluations_scope
228
263
  Evaluation.joins(:agent).where(agent: owner_agents)
229
264
  end
@@ -2,14 +2,22 @@
2
2
 
3
3
  module ActionAgent
4
4
  module Api
5
- # Model catalogs for the agent builder/editor dropdowns.
5
+ # Model catalogs for the dashboard's model pickers: the agent builder and
6
+ # editor, and the evaluation forms.
6
7
  #
7
8
  # Hosted providers get a curated list of current models (kept here, server
8
9
  # side, so the UI can't drift stale). Ollama is queried live from the
9
10
  # account's configured host (Settings -> Provider API Keys, falling back to
10
11
  # the platform config) so locally pulled models appear; OpenRouter is
11
- # queried from its public catalog. Live lookups fall back to the curated
12
- # list on any failure.
12
+ # queried from its public catalog, and Anthropic from its Models API with
13
+ # the account's key. Live lookups fall back to the curated list on any
14
+ # failure.
15
+ #
16
+ # When the host app loads RubyLLM, the chat models its registry lists for
17
+ # the provider that take and return text follow: RubyLLM's bundled
18
+ # catalog, or the host's own model table when it configured one. The list
19
+ # stays de-duplicated, and its first id, the builder's preselected
20
+ # default, is always the live or curated one.
13
21
  class ProviderModelsController < BaseController
14
22
  before_action :require_owner!
15
23
 
@@ -42,7 +50,7 @@ module ActionAgent
42
50
  source = "curated"
43
51
  end
44
52
 
45
- render json: { provider: provider, models: models, source: source }
53
+ render json: { provider: provider, models: (models + registry_models(provider)).uniq, source: source }
46
54
  end
47
55
 
48
56
  private
@@ -106,6 +114,29 @@ module ActionAgent
106
114
  nil
107
115
  end
108
116
 
117
+ # Returns the ids of the chat models RubyLLM's registry lists for
118
+ # +provider+ that take and return text. Empty without RubyLLM, or when
119
+ # the registry raises. This engine's provider names are RubyLLM's own
120
+ # slugs.
121
+ #
122
+ # `chat_models` alone is not enough: RubyLLM counts a model as a chat
123
+ # model whenever its output modalities name no other kind, none at all
124
+ # included. That takes in the speech, transcription, moderation and
125
+ # completion-only models its registry lists with no modalities, and the
126
+ # transcription models it lists as taking audio. The few chat models it
127
+ # lists with no modalities are left out with them.
128
+ def registry_models(provider)
129
+ return [] unless defined?(::RubyLLM) && ::RubyLLM.respond_to?(:models)
130
+
131
+ ::RubyLLM.models.by_provider(provider).chat_models.filter_map do |model|
132
+ modalities = model.modalities
133
+ model.id.presence if Array(modalities&.input).include?("text") && Array(modalities&.output).include?("text")
134
+ end
135
+ rescue StandardError => e
136
+ Rails.logger.warn("[ProviderModels] RubyLLM registry lookup failed: #{e.message}")
137
+ []
138
+ end
139
+
109
140
  def fetch_json(uri)
110
141
  response = Net::HTTP.start(
111
142
  uri.host, uri.port,
@@ -49,6 +49,12 @@ module ActionAgent
49
49
  new(agent_record, run).call
50
50
  end
51
51
 
52
+ # Returns the providers in Agent::PROVIDERS a run on +owner+'s behalf has
53
+ # credentials for, in that order (see #available_providers).
54
+ def self.available_providers(owner)
55
+ new(nil, nil, owner: owner).available_providers
56
+ end
57
+
52
58
  # Tool-call keywords that name the caller. The model's arguments and the
53
59
  # run's actor share one keyword namespace by the time they reach a tool,
54
60
  # so anything a model emits under these names is dropped before the call:
@@ -56,13 +62,23 @@ module ActionAgent
56
62
  # documents a model reads are attacker-reachable.
57
63
  ACTOR_KEYWORDS = %i[actor current_user].freeze
58
64
 
59
- def initialize(agent_record, run)
65
+ # +owner+ is whose provider credentials the run uses: the agent record's
66
+ # owner unless given.
67
+ def initialize(agent_record, run, owner: nil)
60
68
  @agent_record = agent_record
61
69
  @run = run
70
+ @owner = owner
62
71
  @tool_invocations = []
63
72
  @event_sequence = 0
64
73
  end
65
74
 
75
+ # Returns the providers in Agent::PROVIDERS the owner's credentials, or
76
+ # the host's config, let a run use: #provider_available? for each.
77
+ # @return [Array<String>]
78
+ def available_providers
79
+ Agent::PROVIDERS.select { |name| provider_available?(name) }
80
+ end
81
+
66
82
  # The caller this run executes on behalf of, or nil when it runs
67
83
  # unattributed. Passed to every tool as +actor:+ — a host's SchemaTools
68
84
  # scope block, Pundit policy or agent callback decides what that means.
@@ -943,7 +959,7 @@ module ActionAgent
943
959
  # The agent's owner under the configured mode; nil when the install
944
960
  # has no owner model at all.
945
961
  def owner
946
- @owner ||= @agent_record.owner
962
+ @owner ||= @agent_record&.owner
947
963
  end
948
964
 
949
965
  def record_trace(root_span)
@@ -15,8 +15,17 @@ module ActionAgent
15
15
  new(evaluation).call
16
16
  end
17
17
 
18
- def initialize(evaluation)
18
+ # Returns the provider +owner+'s evaluation judge runs on (see
19
+ # #judge_provider), or nil when no provider has credentials.
20
+ def self.judge_provider_for(owner)
21
+ new(nil, owner: owner).judge_provider
22
+ end
23
+
24
+ # +owner+ is whose provider credentials the judge uses: the evaluated
25
+ # agent's owner unless given.
26
+ def initialize(evaluation, owner: nil)
19
27
  @evaluation = evaluation
28
+ @owner = owner
20
29
  end
21
30
 
22
31
  def call
@@ -71,6 +80,17 @@ module ActionAgent
71
80
  raise
72
81
  end
73
82
 
83
+ # Returns the provider the judge runs on: the first of Anthropic, OpenAI
84
+ # and OpenRouter the owner or the host's config has a key for, else Ollama
85
+ # when the owner configured a host; nil when none. The judge runs
86
+ # `judge_model` as that provider's own model id.
87
+ def judge_provider
88
+ @judge_provider ||=
89
+ %i[anthropic openai openrouter].find do |name|
90
+ owner_provider_options(name).any? || global_provider_token?(name)
91
+ end || (:ollama if owner_provider_options(:ollama).any?)
92
+ end
93
+
74
94
  private
75
95
 
76
96
  # Scores each sample-based criterion once per candidate model cohort and
@@ -547,13 +567,6 @@ module ActionAgent
547
567
  judge_provider.present?
548
568
  end
549
569
 
550
- def judge_provider
551
- @judge_provider ||=
552
- %i[anthropic openai openrouter].find do |name|
553
- owner_provider_options(name).any? || global_provider_token?(name)
554
- end || (:ollama if owner_provider_options(:ollama).any?)
555
- end
556
-
557
570
  def global_provider_token?(name)
558
571
  config = ActiveAgent.configuration[name]
559
572
  config.respond_to?(:[]) && config[:access_token].present?
@@ -571,7 +584,7 @@ module ActionAgent
571
584
  end
572
585
 
573
586
  def owner
574
- @owner ||= @evaluation.agent.owner
587
+ @owner ||= @evaluation&.agent&.owner
575
588
  end
576
589
 
577
590
  def judge_class
@@ -3,7 +3,7 @@
3
3
  module ActionAgent
4
4
  # Names the MCP server behind a tool an evaluation run needs, for the
5
5
  # report's fix items (ActiveAgent::Evals::Report#fix_items): a scenario
6
- # that expected +search_slots+ and never got it is fixed by enabling the
6
+ # that expected +track_shipment+ and never got it is fixed by enabling the
7
7
  # server that serves it, and the item can only say so — and deep-link to
8
8
  # MCP Services — when something here can name that server.
9
9
  #
@@ -97,7 +97,7 @@ module ActionAgent
97
97
  end
98
98
 
99
99
  # normalized key => the display name a configured hash entry carries
100
- # alongside its key ({"key" => "booking", "name" => "Booking Service"}).
100
+ # alongside its key ({"key" => "shipping", "name" => "Shipping Desk"}).
101
101
  def configured_names
102
102
  @configured_names ||= configured_entries.each_with_object({}) do |entry, map|
103
103
  next unless entry.respond_to?(:key?)
@@ -111,7 +111,7 @@ module ActionAgent
111
111
  end
112
112
 
113
113
  # bare tool name => server key, from configured entries that list the
114
- # tools they serve ({"name" => "booking", "tools" => ["search_slots"]}),
114
+ # tools they serve ({"name" => "shipping", "tools" => ["track_shipment"]}),
115
115
  # in the catalog's own +tool_hints+ spelling or as tool hashes.
116
116
  def configured_tools
117
117
  @configured_tools ||= configured_entries.each_with_object({}) do |entry, map|
@@ -21,6 +21,11 @@ module ActionAgent
21
21
  class ScenarioEvaluationRunner < EvaluationRunnerService
22
22
  Evals = ActiveAgent::Evals
23
23
 
24
+ # The providers a candidate model name resolves against. `mock` is the
25
+ # framework's test double, accepted so the test suite can compare cohorts
26
+ # offline.
27
+ CANDIDATE_PROVIDERS = (Agent::PROVIDERS + %w[mock]).freeze
28
+
24
29
  # `run` is an EvaluationRun created ahead of time (by run_later!, so the
25
30
  # UI can show it pending while the job waits); absent, one is created here.
26
31
  def self.call(evaluation, selection: {}, run: nil)
@@ -126,15 +131,14 @@ module ActionAgent
126
131
  end
127
132
 
128
133
  # The models to compare: an explicit selection, else the evaluation's
129
- # compare_models, else the agent as configured. `mock` is the framework's
130
- # test double, accepted so the test suite can compare cohorts offline.
134
+ # compare_models, else the agent as configured.
131
135
  def model_specs
132
136
  names = Array(@selection[:models]).presence || @evaluation.compare_models
133
- specs = Evals::ModelSpec.parse_all(names, default_provider: @evaluation.agent.provider, providers: Agent::PROVIDERS + %w[mock])
137
+ specs = Evals::ModelSpec.parse_all(names, default_provider: @evaluation.agent.provider, providers: CANDIDATE_PROVIDERS)
134
138
  # parse_all resolves a bare name against `providers:` but passes through a
135
139
  # `provider/model` whose provider is not in that list, so the run would
136
140
  # otherwise reach the replay with a provider nothing can serve.
137
- unsupported = specs.map(&:provider).uniq - (Agent::PROVIDERS + %w[mock])
141
+ unsupported = specs.map(&:provider).uniq - CANDIDATE_PROVIDERS
138
142
  raise ArgumentError, "unsupported model provider: #{unsupported.to_sentence}" if unsupported.any?
139
143
 
140
144
  return specs if specs.any?
data/config/routes.rb CHANGED
@@ -50,6 +50,9 @@ ActionAgent::Engine.routes.draw do
50
50
  # and a fresh one to pin a first message to.
51
51
  get :conversations
52
52
  post :conversations, action: :create_conversation
53
+ # The evaluation form's models field: the model names this agent's
54
+ # generations were recorded under.
55
+ get :recorded_models
53
56
  end
54
57
  collection do
55
58
  get :presets
@@ -1,5 +1,5 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module ActionAgent
4
- VERSION = "1.7.0"
4
+ VERSION = "1.7.1"
5
5
  end
metadata CHANGED
@@ -1,14 +1,14 @@
1
1
  --- !ruby/object:Gem::Specification
2
2
  name: actionagent
3
3
  version: !ruby/object:Gem::Version
4
- version: 1.7.0
4
+ version: 1.7.1
5
5
  platform: ruby
6
6
  authors:
7
7
  - Justin Bowen
8
8
  autorequire:
9
9
  bindir: bin
10
10
  cert_chain: []
11
- date: 2026-09-24 00:00:00.000000000 Z
11
+ date: 2026-09-30 00:00:00.000000000 Z
12
12
  dependencies:
13
13
  - !ruby/object:Gem::Dependency
14
14
  name: activeagent