actionagent 1.6.4 → 1.7.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (39) hide show
  1. checksums.yaml +4 -4
  2. data/README.md +5 -0
  3. data/app/assets/builds/action_agent.css +1 -1
  4. data/app/assets/builds/action_agent.js +57 -57
  5. data/app/controllers/action_agent/api/agents_controller.rb +28 -3
  6. data/app/controllers/action_agent/api/base_controller.rb +20 -4
  7. data/app/controllers/action_agent/api/dashboard_assistant_controller.rb +0 -11
  8. data/app/controllers/action_agent/api/evaluation_reports_controller.rb +134 -0
  9. data/app/controllers/action_agent/api/evaluations_controller.rb +88 -12
  10. data/app/controllers/action_agent/api/interactions_controller.rb +2 -3
  11. data/app/controllers/action_agent/api/mcp_controller.rb +3 -1
  12. data/app/controllers/action_agent/api/provider_models_controller.rb +39 -5
  13. data/app/controllers/action_agent/api/trace_reports_controller.rb +20 -1
  14. data/app/controllers/action_agent/api/traces_controller.rb +8 -57
  15. data/app/controllers/action_agent/application_controller.rb +4 -0
  16. data/app/controllers/concerns/action_agent/api/ingest_authentication.rb +94 -0
  17. data/app/models/action_agent/agent.rb +71 -3
  18. data/app/models/action_agent/application_record.rb +4 -0
  19. data/app/models/action_agent/evaluation_run.rb +61 -13
  20. data/app/models/action_agent/telemetry_trace.rb +4 -3
  21. data/app/models/concerns/action_agent/ownable.rb +18 -6
  22. data/app/queries/action_agent/metrics_report.rb +1 -1
  23. data/app/services/action_agent/agent_execution_service.rb +67 -2
  24. data/app/services/action_agent/agent_sync.rb +136 -0
  25. data/app/services/action_agent/agent_tool_roster.rb +1 -1
  26. data/app/services/action_agent/evaluation_report_import.rb +743 -0
  27. data/app/services/action_agent/evaluation_runner_service.rb +141 -19
  28. data/app/services/action_agent/evaluation_tool_resolver.rb +3 -3
  29. data/app/services/action_agent/scenario_evaluation_runner.rb +16 -7
  30. data/config/routes.rb +8 -0
  31. data/lib/action_agent/version.rb +1 -1
  32. data/lib/action_agent.rb +110 -13
  33. data/lib/generators/action_agent/install_generator.rb +26 -3
  34. data/lib/generators/action_agent/templates/action_agent.rb.erb +19 -3
  35. data/lib/generators/action_agent/templates/add_agent_releases.rb.erb +18 -13
  36. data/lib/generators/action_agent/templates/add_evaluation_report_identity.rb.erb +54 -0
  37. data/lib/generators/action_agent/templates/create_active_agent_dashboard_tables.rb.erb +34 -0
  38. data/lib/generators/action_agent/templates/ensure_agent_release_columns.rb.erb +51 -0
  39. metadata +8 -2
@@ -15,8 +15,17 @@ module ActionAgent
15
15
  new(evaluation).call
16
16
  end
17
17
 
18
- def initialize(evaluation)
18
+ # Returns the provider +owner+'s evaluation judge runs on (see
19
+ # #judge_provider), or nil when no provider has credentials.
20
+ def self.judge_provider_for(owner)
21
+ new(nil, owner: owner).judge_provider
22
+ end
23
+
24
+ # +owner+ is whose provider credentials the judge uses: the evaluated
25
+ # agent's owner unless given.
26
+ def initialize(evaluation, owner: nil)
19
27
  @evaluation = evaluation
28
+ @owner = owner
20
29
  end
21
30
 
22
31
  def call
@@ -55,6 +64,9 @@ module ActionAgent
55
64
  scores[criterion["key"]] = stats
56
65
  end
57
66
 
67
+ scores["_cohorts"] = cohort_summaries(samples, per_sample_scores) if samples.any?
68
+ scores["_judge_usage"] = judge_usage if judge_usage
69
+
58
70
  run.update!(
59
71
  status: :complete,
60
72
  scores: scores,
@@ -68,6 +80,17 @@ module ActionAgent
68
80
  raise
69
81
  end
70
82
 
83
+ # Returns the provider the judge runs on: the first of Anthropic, OpenAI
84
+ # and OpenRouter the owner or the host's config has a key for, else Ollama
85
+ # when the owner configured a host; nil when none. The judge runs
86
+ # `judge_model` as that provider's own model id.
87
+ def judge_provider
88
+ @judge_provider ||=
89
+ %i[anthropic openai openrouter].find do |name|
90
+ owner_provider_options(name).any? || global_provider_token?(name)
91
+ end || (:ollama if owner_provider_options(:ollama).any?)
92
+ end
93
+
71
94
  private
72
95
 
73
96
  # Scores each sample-based criterion once per candidate model cohort and
@@ -109,6 +132,11 @@ module ActionAgent
109
132
  scores["_verdict"] = verdict if verdict
110
133
  end
111
134
 
135
+ scores["_cohorts"] = active.to_h do |model, samples|
136
+ [ model, cohort_summary(samples, per_model_sample_scores[model] || {}) ]
137
+ end
138
+ scores["_judge_usage"] = judge_usage if judge_usage
139
+
112
140
  run.update!(
113
141
  status: :complete,
114
142
  scores: scores,
@@ -148,6 +176,98 @@ module ActionAgent
148
176
  end
149
177
  end
150
178
 
179
+ # --- Cohort summaries ------------------------------------------------------
180
+ #
181
+ # Stored under scores["_cohorts"], keyed by model: how many generations
182
+ # were sampled under it, how many cleared every criterion, their latency
183
+ # and token usage, and what those generations cost to serve. A comparison
184
+ # run gets one entry per requested cohort that had generations; a plain
185
+ # run one per model that happened to be among the sampled generations.
186
+ # This is what lets the dashboard show a run as "5/12 passed" per model,
187
+ # the way a scenario run's `_models` summaries do, without re-reading the
188
+ # generations.
189
+ #
190
+ # The cost here is the agent's: what the sampled interactions cost to
191
+ # operate. It is not what this run spent — scoring recorded generations
192
+ # costs nothing until a judge is asked, and the judge's own spend is
193
+ # recorded apart, under "_judge_usage" (see #judge_usage).
194
+
195
+ def cohort_summaries(samples, per_sample_scores)
196
+ samples
197
+ .group_by { |generation| generation.model.presence || "unknown" }
198
+ .transform_values { |group| cohort_summary(group, per_sample_scores) }
199
+ end
200
+
201
+ def cohort_summary(samples, per_sample_scores)
202
+ durations_ms = samples.filter_map do |generation|
203
+ seconds = generation.duration_seconds.to_f
204
+ seconds * 1000 if seconds.positive?
205
+ end
206
+ providers = samples.filter_map { |generation| generation.provider.presence }
207
+ input_tokens = samples.sum { |generation| generation.input_tokens.to_i }
208
+ output_tokens = samples.sum { |generation| generation.output_tokens.to_i }
209
+ costs = samples.filter_map do |generation|
210
+ ModelPricing.estimate(model: generation.model, input_tokens: generation.input_tokens, output_tokens: generation.output_tokens)
211
+ end
212
+
213
+ {
214
+ "samples" => samples.size,
215
+ "passed" => passed_count(per_sample_scores.slice(*samples.map(&:id))),
216
+ "provider" => providers.tally.max_by { |_provider, count| count }&.first,
217
+ "avg_duration_ms" => durations_ms.any? ? (durations_ms.sum / durations_ms.size).round : nil,
218
+ "input_tokens" => input_tokens,
219
+ "output_tokens" => output_tokens,
220
+ "cost" => costs.any? ? costs.sum.round(6) : nil
221
+ }
222
+ end
223
+
224
+ # --- Judge usage -----------------------------------------------------------
225
+ #
226
+ # Every call the judge makes is metered here, apart from the agent's own
227
+ # spend: scoring answers, recommending fixes, writing the verdict and,
228
+ # for a judge_defined evaluation, authoring the KPIs. The agent's cost is
229
+ # the operating figure — what serving these interactions costs — while
230
+ # the judge's is the evaluation's own, offline, agent-to-agent overhead,
231
+ # and a run that reported the two as one number would overstate the
232
+ # first. Persisted as scores["_judge_usage"]:
233
+ #
234
+ # { "calls", "input_tokens", "output_tokens", "cost", "model",
235
+ # "by_kind" => { "score" => n, "recommend" => n, "verdict" => n, "define" => n } }
236
+ #
237
+ # nil until the judge has been asked something, so a rules-only run
238
+ # records no judge at all rather than a judge that cost nothing.
239
+
240
+ def judge_usage
241
+ return nil if @judge_usage.nil?
242
+
243
+ @judge_usage.merge("cost" => @judge_usage["cost"]&.round(6))
244
+ end
245
+
246
+ # Asks the judge and meters the answer. `kind` is the call's purpose —
247
+ # :score, :recommend, :verdict or :define — the same vocabulary
248
+ # ActiveAgent::Evals::Judge hands a block that accepts `kind:`.
249
+ def judge_generate(kind, message:, instructions:)
250
+ response = judge_class.prompt(message: message, instructions: instructions).generate_now
251
+ record_judge_call(kind, response)
252
+ response
253
+ end
254
+
255
+ def record_judge_call(kind, response)
256
+ usage = response.respond_to?(:usage) ? response.usage : nil
257
+ input_tokens = usage.respond_to?(:input_tokens) ? usage.input_tokens.to_i : 0
258
+ output_tokens = usage.respond_to?(:output_tokens) ? usage.output_tokens.to_i : 0
259
+ model = (response.respond_to?(:model) && response.model.presence) || @evaluation.judge_model.presence
260
+ cost = ModelPricing.estimate(model: model, input_tokens: input_tokens, output_tokens: output_tokens)
261
+
262
+ @judge_usage ||= { "calls" => 0, "input_tokens" => 0, "output_tokens" => 0, "cost" => nil, "model" => nil, "by_kind" => {} }
263
+ @judge_usage["calls"] += 1
264
+ @judge_usage["input_tokens"] += input_tokens
265
+ @judge_usage["output_tokens"] += output_tokens
266
+ @judge_usage["cost"] = (@judge_usage["cost"] || 0.0) + cost if cost
267
+ @judge_usage["model"] ||= model
268
+ @judge_usage["by_kind"][kind.to_s] = @judge_usage["by_kind"].fetch(kind.to_s, 0) + 1
269
+ end
270
+
151
271
  def sample_generations(model: nil)
152
272
  scope = @evaluation.agent.generations
153
273
  scope = scope.where(model: model) if model
@@ -175,7 +295,7 @@ module ActionAgent
175
295
  if total.zero?
176
296
  return {
177
297
  "skipped" => true,
178
- "reason" => "No telemetry traces for #{@evaluation.agent.telemetry_agent_class} in the last #{window_hours}h"
298
+ "reason" => "No telemetry traces for #{telemetry_source} in the last #{window_hours}h"
179
299
  }
180
300
  end
181
301
 
@@ -213,12 +333,18 @@ module ActionAgent
213
333
  end
214
334
 
215
335
  def telemetry_traces(window_hours)
216
- ActionAgent.trace_model
217
- .for_account(ActionAgent.tenant_for(owner))
218
- .for_agent(@evaluation.agent.telemetry_agent_class)
336
+ @evaluation.agent
337
+ .telemetry_traces(ActionAgent.trace_model.for_account(ActionAgent.tenant_for(owner)))
219
338
  .for_date_range(window_hours.hours.ago, Time.current)
220
339
  end
221
340
 
341
+ # What a skip reason says was looked for: the observed agent itself, or
342
+ # the class any other agent's traces are reported under.
343
+ def telemetry_source
344
+ agent = @evaluation.agent
345
+ agent.observed? ? agent.name : agent.telemetry_agent_class
346
+ end
347
+
222
348
  # Returns 0.0..1.0, or nil when the criterion cannot be scored.
223
349
  def score_sample(criterion, generation)
224
350
  config = criterion["config"] || {}
@@ -273,10 +399,11 @@ module ActionAgent
273
399
  raise "Judge-defined KPIs need provider credentials (add a provider API key in Settings)"
274
400
  end
275
401
 
276
- response = judge_class.prompt(
402
+ response = judge_generate(
403
+ :define,
277
404
  message: kpi_definition_prompt,
278
405
  instructions: "You define measurable evaluation KPIs for AI agents. Respond ONLY with JSON."
279
- ).generate_now
406
+ )
280
407
 
281
408
  kpis = parse_kpis(response.message&.content)
282
409
  raise "Judge returned no usable KPIs — try again or add criteria manually" if kpis.empty?
@@ -354,7 +481,8 @@ module ActionAgent
354
481
  "#{key}: #{cells.join(', ')}"
355
482
  end
356
483
 
357
- response = judge_class.prompt(
484
+ response = judge_generate(
485
+ :verdict,
358
486
  message: <<~PROMPT,
359
487
  An AI agent was evaluated under multiple models. Its goals:
360
488
  ---
@@ -368,7 +496,7 @@ module ActionAgent
368
496
  Respond ONLY with JSON: {"winner": "<model>", "rationale": "<at most two sentences>"}
369
497
  PROMPT
370
498
  instructions: "You are an impartial evaluation judge comparing model cohorts. Respond ONLY with JSON."
371
- ).generate_now
499
+ )
372
500
 
373
501
  json = response.message&.content.to_s[/\{.*\}/m]
374
502
  verdict = json ? JSON.parse(json) : nil
@@ -390,10 +518,11 @@ module ActionAgent
390
518
  return nil unless judge_available?
391
519
  return nil if generation.content.blank?
392
520
 
393
- response = judge_class.prompt(
521
+ response = judge_generate(
522
+ :score,
394
523
  message: judge_prompt(criterion, generation),
395
524
  instructions: "You are an impartial evaluation judge. Respond ONLY with JSON: {\"score\": <float between 0.0 and 1.0>}"
396
- ).generate_now
525
+ )
397
526
 
398
527
  parse_judge_score(response.message&.content)
399
528
  rescue StandardError => e
@@ -438,13 +567,6 @@ module ActionAgent
438
567
  judge_provider.present?
439
568
  end
440
569
 
441
- def judge_provider
442
- @judge_provider ||=
443
- %i[anthropic openai openrouter].find do |name|
444
- owner_provider_options(name).any? || global_provider_token?(name)
445
- end || (:ollama if owner_provider_options(:ollama).any?)
446
- end
447
-
448
570
  def global_provider_token?(name)
449
571
  config = ActiveAgent.configuration[name]
450
572
  config.respond_to?(:[]) && config[:access_token].present?
@@ -462,7 +584,7 @@ module ActionAgent
462
584
  end
463
585
 
464
586
  def owner
465
- @owner ||= @evaluation.agent.owner
587
+ @owner ||= @evaluation&.agent&.owner
466
588
  end
467
589
 
468
590
  def judge_class
@@ -3,7 +3,7 @@
3
3
  module ActionAgent
4
4
  # Names the MCP server behind a tool an evaluation run needs, for the
5
5
  # report's fix items (ActiveAgent::Evals::Report#fix_items): a scenario
6
- # that expected +search_slots+ and never got it is fixed by enabling the
6
+ # that expected +track_shipment+ and never got it is fixed by enabling the
7
7
  # server that serves it, and the item can only say so — and deep-link to
8
8
  # MCP Services — when something here can name that server.
9
9
  #
@@ -97,7 +97,7 @@ module ActionAgent
97
97
  end
98
98
 
99
99
  # normalized key => the display name a configured hash entry carries
100
- # alongside its key ({"key" => "booking", "name" => "Booking Service"}).
100
+ # alongside its key ({"key" => "shipping", "name" => "Shipping Desk"}).
101
101
  def configured_names
102
102
  @configured_names ||= configured_entries.each_with_object({}) do |entry, map|
103
103
  next unless entry.respond_to?(:key?)
@@ -111,7 +111,7 @@ module ActionAgent
111
111
  end
112
112
 
113
113
  # bare tool name => server key, from configured entries that list the
114
- # tools they serve ({"name" => "booking", "tools" => ["search_slots"]}),
114
+ # tools they serve ({"name" => "shipping", "tools" => ["track_shipment"]}),
115
115
  # in the catalog's own +tool_hints+ spelling or as tool hashes.
116
116
  def configured_tools
117
117
  @configured_tools ||= configured_entries.each_with_object({}) do |entry, map|
@@ -21,6 +21,11 @@ module ActionAgent
21
21
  class ScenarioEvaluationRunner < EvaluationRunnerService
22
22
  Evals = ActiveAgent::Evals
23
23
 
24
+ # The providers a candidate model name resolves against. `mock` is the
25
+ # framework's test double, accepted so the test suite can compare cohorts
26
+ # offline.
27
+ CANDIDATE_PROVIDERS = (Agent::PROVIDERS + %w[mock]).freeze
28
+
24
29
  # `run` is an EvaluationRun created ahead of time (by run_later!, so the
25
30
  # UI can show it pending while the job waits); absent, one is created here.
26
31
  def self.call(evaluation, selection: {}, run: nil)
@@ -126,15 +131,14 @@ module ActionAgent
126
131
  end
127
132
 
128
133
  # The models to compare: an explicit selection, else the evaluation's
129
- # compare_models, else the agent as configured. `mock` is the framework's
130
- # test double, accepted so the test suite can compare cohorts offline.
134
+ # compare_models, else the agent as configured.
131
135
  def model_specs
132
136
  names = Array(@selection[:models]).presence || @evaluation.compare_models
133
- specs = Evals::ModelSpec.parse_all(names, default_provider: @evaluation.agent.provider, providers: Agent::PROVIDERS + %w[mock])
137
+ specs = Evals::ModelSpec.parse_all(names, default_provider: @evaluation.agent.provider, providers: CANDIDATE_PROVIDERS)
134
138
  # parse_all resolves a bare name against `providers:` but passes through a
135
139
  # `provider/model` whose provider is not in that list, so the run would
136
140
  # otherwise reach the replay with a provider nothing can serve.
137
- unsupported = specs.map(&:provider).uniq - (Agent::PROVIDERS + %w[mock])
141
+ unsupported = specs.map(&:provider).uniq - CANDIDATE_PROVIDERS
138
142
  raise ArgumentError, "unsupported model provider: #{unsupported.to_sentence}" if unsupported.any?
139
143
 
140
144
  return specs if specs.any?
@@ -261,6 +265,10 @@ module ActionAgent
261
265
  scores["_selection"] = run.selection
262
266
  scores["_metadata"] = report.metadata
263
267
  scores["_judge_label"] = report.judge_label || report.judge&.label
268
+ # The judge's own spend, apart from the replays' (which the results
269
+ # carry and EvaluationRun#usage sums). A host adapter that runs its own
270
+ # judge is out of reach of this meter, so its runs record none.
271
+ scores["_judge_usage"] = judge_usage if judge_usage
264
272
  scores
265
273
  end
266
274
 
@@ -318,12 +326,13 @@ module ActionAgent
318
326
 
319
327
  # The judge the evaluation's owner has credentials for, wrapped for the
320
328
  # evaluation core; nil when none is configured, in which case scoring
321
- # stays on rules and expectations.
329
+ # stays on rules and expectations. The block takes `kind:` so each call
330
+ # is metered under what it was for (EvaluationRunnerService#judge_generate).
322
331
  def evals_judge
323
332
  return nil unless judge_available?
324
333
 
325
- @evals_judge ||= Evals::Judge.new(label: @evaluation.judge_model.presence || judge_provider.to_s) do |instructions:, prompt:|
326
- judge_class.prompt(message: prompt, instructions: instructions).generate_now.message&.content
334
+ @evals_judge ||= Evals::Judge.new(label: @evaluation.judge_model.presence || judge_provider.to_s) do |instructions:, prompt:, kind:|
335
+ judge_generate(kind, message: prompt, instructions: instructions).message&.content
327
336
  end
328
337
  end
329
338
  end
data/config/routes.rb CHANGED
@@ -27,6 +27,11 @@ ActionAgent::Engine.routes.draw do
27
27
  # Authenticated with a bearer token, not a session.
28
28
  resources :traces, only: [ :create ]
29
29
 
30
+ # The collector for evaluation reports an application ran itself
31
+ # (ActiveAgent::Evals::Publisher), at <mount>/api/evaluation_reports.
32
+ # Authenticated like trace ingest, with a bearer token.
33
+ resources :evaluation_reports, only: [ :create ]
34
+
30
35
  # A JSON API has no :new or :edit forms to serve.
31
36
  resources :agents, except: [ :new, :edit ] do
32
37
  member do
@@ -45,6 +50,9 @@ ActionAgent::Engine.routes.draw do
45
50
  # and a fresh one to pin a first message to.
46
51
  get :conversations
47
52
  post :conversations, action: :create_conversation
53
+ # The evaluation form's models field: the model names this agent's
54
+ # generations were recorded under.
55
+ get :recorded_models
48
56
  end
49
57
  collection do
50
58
  get :presets
@@ -1,5 +1,5 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module ActionAgent
4
- VERSION = "1.6.4"
4
+ VERSION = "1.7.1"
5
5
  end
data/lib/action_agent.rb CHANGED
@@ -209,24 +209,77 @@ module ActionAgent
209
209
  # @return [Object, nil] Object responding to #signed_url_for and #fetch_snapshot
210
210
  attr_accessor :storage_service
211
211
 
212
- # Bearer token required by the ingest API in single-tenant mode. When
213
- # unset the local ingest endpoint accepts unauthenticated posts, so set
214
- # it whenever the mount is reachable beyond your own machine.
215
- # (Multi-tenant mode authenticates per-account keys instead.)
212
+ # Bearer token required in single-tenant mode by the endpoints other
213
+ # applications post to: trace ingest (<mount>/api/traces) and published
214
+ # evaluation reports (<mount>/api/evaluation_reports). When unset both
215
+ # accept unauthenticated posts, so set it whenever the mount is reachable
216
+ # beyond your own machine. Trace ingest also takes a form post, which any
217
+ # web page open in a browser on that machine can send it; the report
218
+ # collector takes only application/json, which a page cannot send
219
+ # cross-site. (Multi-tenant mode authenticates per-account keys instead.)
216
220
  # @return [String, nil]
217
221
  attr_accessor :ingest_api_key
218
222
 
223
+ # Concerns included into ActionAgent::ApplicationRecord as it loads, and
224
+ # through it into every engine model. An entry is a Module or the name
225
+ # of one. A name is resolved when the class
226
+ # loads, so an initializer can refer to a constant the host has not
227
+ # autoloaded yet, and a name that resolves to nothing raises NameError
228
+ # there rather than being skipped.
229
+ #
230
+ # ActionAgent.configure do |config|
231
+ # config.model_concerns = ["MyApp::ConnectionSwitching"]
232
+ # end
233
+ #
234
+ # The class loads after the initializers have run, so set this in an
235
+ # initializer; a concern added later is not applied.
236
+ # @return [Array<Module, String>]
237
+ attr_accessor :model_concerns
238
+
239
+ # Concerns included into ActionAgent::ApplicationController as it loads,
240
+ # and through it into every dashboard controller: the React dashboard
241
+ # and its JSON API, the server-rendered console and the MCP facade.
242
+ # They are included ahead of the engine's own callbacks, so a concern's
243
+ # before_action or around_action runs before the dashboard
244
+ # authenticates. Entries are Modules or names, as for model_concerns.
245
+ #
246
+ # Not the endpoints other applications post to: Api::TracesController
247
+ # and Api::EvaluationReportsController authenticate with a bearer token
248
+ # and inherit ActionController::API.
249
+ #
250
+ # ActionAgent.configure do |config|
251
+ # config.controller_concerns = ["MyApp::RequestTagging"]
252
+ # end
253
+ # @return [Array<Module, String>]
254
+ attr_accessor :controller_concerns
255
+
219
256
  # @deprecated Never consumed — dashboard controllers inherit
220
- # ActionController::Base. Retained as a no-op so existing
221
- # initializers that set it keep booting; remove in the next major.
257
+ # ActionController::Base, and controller_concerns is how a host puts
258
+ # its own behaviour on them. Assigning it warns and stores a value
259
+ # nothing reads; removed in the next major.
222
260
  # @return [String]
223
- attr_accessor :base_controller_class
261
+ attr_reader :base_controller_class
262
+
263
+ def base_controller_class=(value)
264
+ deprecator.warn(
265
+ "ActionAgent.base_controller_class has never been consumed and is removed in 2.0. " \
266
+ "Set ActionAgent.controller_concerns to extend the dashboard's controllers."
267
+ )
268
+ @base_controller_class = value
269
+ end
224
270
 
225
- # Called before each run/trace-ingest to enforce host-app limits.
226
- # Receives (owner, kind) where kind is :execution or :trace_ingest, and
227
- # returns nil to allow, or to deny: a message String, or a Hash merged
228
- # into the response so the app can surface its own usage numbers.
229
- # Denials surface as HTTP 402 (execution) / 429 (ingest).
271
+ # Called before each metered action to enforce host-app limits.
272
+ # Receives (owner, kind) and returns nil to allow, or to deny: a message
273
+ # String, or a Hash merged into the response so the app can surface its
274
+ # own usage numbers. The kinds, and how a denial surfaces:
275
+ #
276
+ # :execution — an agent run; HTTP 402
277
+ # :trace_ingest — a POST to <mount>/api/traces; HTTP 429
278
+ # :evaluation_report — a report <mount>/api/evaluation_reports would
279
+ # store (never an identical retry); HTTP 429
280
+ #
281
+ # The owner of an ingest kind is the tenant the key resolved to, nil on a
282
+ # single-tenant install.
230
283
  #
231
284
  # Unset means unlimited, which is what a self-hosted install wants.
232
285
  # @return [Proc, nil]
@@ -256,6 +309,21 @@ module ActionAgent
256
309
  # @return [Boolean]
257
310
  attr_accessor :execution_enabled
258
311
 
312
+ # Whether a run of an agent that mirrors a host class executes that class,
313
+ # instead of the class the engine builds from the record's `tools` and
314
+ # `instructions` columns.
315
+ #
316
+ # Off by default: it changes what a run of a mirrored agent executes, and
317
+ # a host that has tuned its dashboard records around the dynamic runtime
318
+ # should opt in deliberately. Dashboard-authored agents — the ones with no
319
+ # `agent_class_name` — are unaffected either way.
320
+ #
321
+ # On, a mirrored agent runs its real tools, delegations and instructions,
322
+ # so an evaluation scores the agent production runs rather than a
323
+ # flattened copy of it.
324
+ # @return [Boolean]
325
+ attr_accessor :run_host_agent_classes
326
+
259
327
  # Whether the "Ask ActiveAgents" assistant is available.
260
328
  #
261
329
  # The assistant is a tool for developing and CI-ing agents: it sends
@@ -314,7 +382,9 @@ module ActionAgent
314
382
 
315
383
  # Called after the dashboard performs a metered action, as
316
384
  # (owner, kind) — the counterpart to quota_checker, for host apps that
317
- # track usage against a plan. Unset means nothing is counted.
385
+ # track usage against a plan. The kinds are :execution, for each agent
386
+ # run, and :evaluation_report, for each report the collector stores; an
387
+ # identical retry is not counted again. Unset means nothing is counted.
318
388
  # @return [Proc, nil]
319
389
  attr_accessor :usage_recorder
320
390
 
@@ -323,6 +393,11 @@ module ActionAgent
323
393
  # nobody in single-tenant mode. A host app whose agents hang off a
324
394
  # different record (the platform's hang off the account's owning user)
325
395
  # supplies its own mapping.
396
+ #
397
+ # A published evaluation report's agent is placed the same way: the
398
+ # resolver receives an unsaved trace with the publishing tenant as its
399
+ # account, and the report's source and agent name as its service_name and
400
+ # agent_class. In multi-tenant mode it must not return nil there.
326
401
  # @return [Proc, nil]
327
402
  attr_accessor :trace_owner_resolver
328
403
 
@@ -513,6 +588,19 @@ module ActionAgent
513
588
  name&.safe_constantize
514
589
  end
515
590
 
591
+ # The modules model_concerns names. ApplicationRecord reads it as it loads.
592
+ # @return [Array<Module>]
593
+ def model_concern_modules
594
+ resolve_concerns(model_concerns)
595
+ end
596
+
597
+ # The modules controller_concerns names. ApplicationController reads it
598
+ # as it loads.
599
+ # @return [Array<Module>]
600
+ def controller_concern_modules
601
+ resolve_concerns(controller_concerns)
602
+ end
603
+
516
604
  # Configures the dashboard.
517
605
  #
518
606
  # @yield [config] Configuration block
@@ -540,10 +628,13 @@ module ActionAgent
540
628
  @storage_service = nil
541
629
  @ingest_api_key = nil
542
630
  @base_controller_class = "ActionController::Base" # deprecated no-op
631
+ @model_concerns = []
632
+ @controller_concerns = []
543
633
  @quota_checker = nil
544
634
  @provider_credentials_resolver = nil
545
635
  @sandbox_backends = {}
546
636
  @execution_enabled = true
637
+ @run_host_agent_classes = false
547
638
  @assistant_enabled = nil
548
639
 
549
640
  @scenario_evaluation_adapter_resolver = nil
@@ -639,6 +730,12 @@ module ActionAgent
639
730
  def schema_tool_class_for(name)
640
731
  schema_tool_classes.find { |klass| klass.tool?(name) }
641
732
  end
733
+
734
+ private
735
+
736
+ def resolve_concerns(entries)
737
+ Array(entries).map { |entry| entry.is_a?(Module) ? entry : entry.to_s.constantize }
738
+ end
642
739
  end
643
740
 
644
741
  # Set defaults
@@ -77,13 +77,36 @@ module ActionAgent
77
77
  )
78
78
  end
79
79
 
80
+ # add_agent_releases is emitted before the dashboard tables, so on an
81
+ # install generated fresh before the create-table migration carried the
82
+ # release columns it skipped every table, and its copy may name tables
83
+ # without the configured prefix. This adds whatever is still missing,
84
+ # and changes nothing where the columns exist.
85
+ unless existing_migration?("ensure_agent_release_columns")
86
+ migration_template(
87
+ "ensure_agent_release_columns.rb.erb",
88
+ "db/migrate/ensure_agent_release_columns.rb"
89
+ )
90
+ end
91
+
80
92
  # Scenario suites arrived after the dashboard tables shipped, so an
81
93
  # install that already has those still needs this one.
82
- return if existing_migration?("create_active_agent_evaluation_scenarios")
94
+ unless existing_migration?("create_active_agent_evaluation_scenarios")
95
+ migration_template(
96
+ "create_active_agent_evaluation_scenarios.rb.erb",
97
+ "db/migrate/create_active_agent_evaluation_scenarios.rb"
98
+ )
99
+ end
100
+
101
+ # Published evaluation reports arrived after the dashboard tables
102
+ # shipped. Emitted after them, so on a fresh install it runs once the
103
+ # table exists, finds the columns the create-table migration made, and
104
+ # changes nothing.
105
+ return if existing_migration?("add_evaluation_report_identity")
83
106
 
84
107
  migration_template(
85
- "create_active_agent_evaluation_scenarios.rb.erb",
86
- "db/migrate/create_active_agent_evaluation_scenarios.rb"
108
+ "add_evaluation_report_identity.rb.erb",
109
+ "db/migrate/add_evaluation_report_identity.rb"
87
110
  )
88
111
  end
89
112
 
@@ -44,9 +44,11 @@ ActionAgent.configure do |config|
44
44
  # Ingest API authentication
45
45
  # ==========================================================================
46
46
  #
47
- # Bearer token other apps must send when posting traces to the mounted
48
- # ingest endpoint (/activeagents/api/traces). Leave unset only when the
49
- # mount is not reachable beyond your own machine.
47
+ # Bearer token other apps must send when posting traces
48
+ # (/activeagents/api/traces) or evaluation reports
49
+ # (/activeagents/api/evaluation_reports) to the mount. Leave unset only
50
+ # when the mount is not reachable beyond your own machine; even then, any web
51
+ # page open in a browser there can post form data to the trace endpoint.
50
52
  #
51
53
  # config.ingest_api_key = Rails.application.credentials.dig(:active_agent, :ingest_api_key)
52
54
 
@@ -78,6 +80,20 @@ ActionAgent.configure do |config|
78
80
  # nothing rather than everything. An empty dashboard for a signed-in user
79
81
  # means the resolver above returned nil.
80
82
 
83
+ # ==========================================================================
84
+ # Host integration
85
+ # ==========================================================================
86
+ #
87
+ # Concerns your own models and controllers carry, applied to the engine's.
88
+ # model_concerns are included into ActionAgent::ApplicationRecord, and so
89
+ # into every engine model. controller_concerns are included into
90
+ # ActionAgent::ApplicationController ahead of its own callbacks, and so
91
+ # into every dashboard controller — not the bearer-token ingest endpoints.
92
+ # Modules or their names; a name is resolved when the class loads.
93
+ #
94
+ # config.model_concerns = ["ConnectionSwitching"]
95
+ # config.controller_concerns = ["RequestTagging"]
96
+
81
97
  # ==========================================================================
82
98
  # Ask ActiveAgents assistant
83
99
  # ==========================================================================