actionagent 1.6.4 → 1.7.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +5 -0
- data/app/assets/builds/action_agent.css +1 -1
- data/app/assets/builds/action_agent.js +57 -57
- data/app/controllers/action_agent/api/agents_controller.rb +28 -3
- data/app/controllers/action_agent/api/base_controller.rb +20 -4
- data/app/controllers/action_agent/api/dashboard_assistant_controller.rb +0 -11
- data/app/controllers/action_agent/api/evaluation_reports_controller.rb +134 -0
- data/app/controllers/action_agent/api/evaluations_controller.rb +88 -12
- data/app/controllers/action_agent/api/interactions_controller.rb +2 -3
- data/app/controllers/action_agent/api/mcp_controller.rb +3 -1
- data/app/controllers/action_agent/api/provider_models_controller.rb +39 -5
- data/app/controllers/action_agent/api/trace_reports_controller.rb +20 -1
- data/app/controllers/action_agent/api/traces_controller.rb +8 -57
- data/app/controllers/action_agent/application_controller.rb +4 -0
- data/app/controllers/concerns/action_agent/api/ingest_authentication.rb +94 -0
- data/app/models/action_agent/agent.rb +71 -3
- data/app/models/action_agent/application_record.rb +4 -0
- data/app/models/action_agent/evaluation_run.rb +61 -13
- data/app/models/action_agent/telemetry_trace.rb +4 -3
- data/app/models/concerns/action_agent/ownable.rb +18 -6
- data/app/queries/action_agent/metrics_report.rb +1 -1
- data/app/services/action_agent/agent_execution_service.rb +67 -2
- data/app/services/action_agent/agent_sync.rb +136 -0
- data/app/services/action_agent/agent_tool_roster.rb +1 -1
- data/app/services/action_agent/evaluation_report_import.rb +743 -0
- data/app/services/action_agent/evaluation_runner_service.rb +141 -19
- data/app/services/action_agent/evaluation_tool_resolver.rb +3 -3
- data/app/services/action_agent/scenario_evaluation_runner.rb +16 -7
- data/config/routes.rb +8 -0
- data/lib/action_agent/version.rb +1 -1
- data/lib/action_agent.rb +110 -13
- data/lib/generators/action_agent/install_generator.rb +26 -3
- data/lib/generators/action_agent/templates/action_agent.rb.erb +19 -3
- data/lib/generators/action_agent/templates/add_agent_releases.rb.erb +18 -13
- data/lib/generators/action_agent/templates/add_evaluation_report_identity.rb.erb +54 -0
- data/lib/generators/action_agent/templates/create_active_agent_dashboard_tables.rb.erb +34 -0
- data/lib/generators/action_agent/templates/ensure_agent_release_columns.rb.erb +51 -0
- metadata +8 -2
|
@@ -18,6 +18,8 @@ module ActionAgent
|
|
|
18
18
|
DEFAULT_LIST_SORT = "recent"
|
|
19
19
|
# Conversations returned to the runner's picker when no limit is asked for.
|
|
20
20
|
CONVERSATIONS_LIMIT = 50
|
|
21
|
+
# Model names returned by #recorded_models.
|
|
22
|
+
RECORDED_MODELS_LIMIT = 50
|
|
21
23
|
# Keywords Agent#execute takes in its own right, which per-run overrides
|
|
22
24
|
# must never supply (see #execution_params). `actor` is here for the
|
|
23
25
|
# same reason as the rest and one more: a keyword splat wins over the
|
|
@@ -27,7 +29,7 @@ module ActionAgent
|
|
|
27
29
|
|
|
28
30
|
before_action :set_agent, only: [
|
|
29
31
|
:show, :update, :destroy, :versions, :runs, :execute, :test, :restore, :duplicate, :export, :analytics,
|
|
30
|
-
:tool_roster, :conversations, :create_conversation
|
|
32
|
+
:tool_roster, :conversations, :create_conversation, :recorded_models
|
|
31
33
|
]
|
|
32
34
|
before_action :require_execution_enabled!, only: [ :execute, :test ]
|
|
33
35
|
before_action :require_owner!, only: [ :execute, :test ]
|
|
@@ -90,7 +92,7 @@ module ActionAgent
|
|
|
90
92
|
if @agent.save
|
|
91
93
|
render json: { agent: agent_json(@agent, include_details: true) }, status: :created
|
|
92
94
|
else
|
|
93
|
-
render json:
|
|
95
|
+
render json: agent_errors_json(@agent), status: :unprocessable_entity
|
|
94
96
|
end
|
|
95
97
|
end
|
|
96
98
|
|
|
@@ -99,7 +101,7 @@ module ActionAgent
|
|
|
99
101
|
if @agent.update(agent_params)
|
|
100
102
|
render json: { agent: agent_json(@agent, include_details: true) }
|
|
101
103
|
else
|
|
102
|
-
render json:
|
|
104
|
+
render json: agent_errors_json(@agent), status: :unprocessable_entity
|
|
103
105
|
end
|
|
104
106
|
end
|
|
105
107
|
|
|
@@ -274,6 +276,23 @@ module ActionAgent
|
|
|
274
276
|
).as_json
|
|
275
277
|
end
|
|
276
278
|
|
|
279
|
+
# GET /api/agents/:id/recorded_models
|
|
280
|
+
#
|
|
281
|
+
# The model names this agent's generations were recorded under, most
|
|
282
|
+
# recently used first. An evaluation without scenarios compares the
|
|
283
|
+
# generations recorded under each name it is given, so these are the
|
|
284
|
+
# names its models field suggests.
|
|
285
|
+
def recorded_models
|
|
286
|
+
generations = AgentGeneration.arel_table
|
|
287
|
+
models = @agent.generations.where.not(model: [ nil, "" ])
|
|
288
|
+
.group(generations[:model])
|
|
289
|
+
.order(Arel::Nodes::Descending.new(generations[:created_at].maximum))
|
|
290
|
+
.limit(RECORDED_MODELS_LIMIT)
|
|
291
|
+
.pluck(generations[:model])
|
|
292
|
+
|
|
293
|
+
render json: { models: models }
|
|
294
|
+
end
|
|
295
|
+
|
|
277
296
|
# GET /api/agents/:id/analytics
|
|
278
297
|
#
|
|
279
298
|
# Every execution of this agent, whoever ran it — the same merged model
|
|
@@ -495,6 +514,12 @@ module ActionAgent
|
|
|
495
514
|
@agent = owner_agents.find(params[:id])
|
|
496
515
|
end
|
|
497
516
|
|
|
517
|
+
# +errors+ for a form-level summary; +field_errors+ (attribute => full
|
|
518
|
+
# messages) so the builder and editor can put each under its field.
|
|
519
|
+
def agent_errors_json(agent)
|
|
520
|
+
{ errors: agent.errors.full_messages, field_errors: agent.errors.to_hash(true) }
|
|
521
|
+
end
|
|
522
|
+
|
|
498
523
|
def agent_params
|
|
499
524
|
permitted = params.require(:agent).permit(
|
|
500
525
|
:name, :description, :provider, :model, :instructions,
|
|
@@ -12,11 +12,23 @@ module ActionAgent
|
|
|
12
12
|
# install, a per-user install and a multi-tenant platform all read the
|
|
13
13
|
# same controllers.
|
|
14
14
|
#
|
|
15
|
-
#
|
|
16
|
-
#
|
|
15
|
+
# Because it authenticates with the host's session cookie, it keeps the
|
|
16
|
+
# forgery protection ApplicationController turns on: the dashboard sends
|
|
17
|
+
# the page's CSRF token with every mutating request (frontend
|
|
18
|
+
# utils/apiFetch.mjs). Endpoints that authenticate with a bearer token
|
|
19
|
+
# instead — the telemetry ingest endpoint (Api::TracesController), the
|
|
20
|
+
# evaluation report collector (Api::EvaluationReportsController) and the
|
|
21
|
+
# MCP facade (Api::MCPController) — are exempt.
|
|
17
22
|
class BaseController < ActionAgent::ApplicationController
|
|
18
|
-
|
|
19
|
-
|
|
23
|
+
# Rails 8.2 verifies forgery protection from the browser's Sec-Fetch-Site
|
|
24
|
+
# header, renamed the failure to InvalidCrossOriginRequest, and deprecated
|
|
25
|
+
# the old name. Rescue whichever names the running Rails defines, so a
|
|
26
|
+
# rejected request answers with the dashboard's JSON either way.
|
|
27
|
+
# const_defined? does not fire the deprecation the bare constant would.
|
|
28
|
+
rescue_from ActionController::InvalidCrossOriginRequest, with: :invalid_authenticity_token
|
|
29
|
+
if ActionController.const_defined?(:InvalidAuthenticityToken, false)
|
|
30
|
+
rescue_from ActionController::InvalidAuthenticityToken, with: :invalid_authenticity_token
|
|
31
|
+
end
|
|
20
32
|
rescue_from ActiveRecord::RecordNotFound, with: :not_found
|
|
21
33
|
rescue_from ActiveRecord::RecordInvalid, with: :unprocessable_entity
|
|
22
34
|
rescue_from ActionController::ParameterMissing, with: :bad_request
|
|
@@ -24,6 +36,10 @@ module ActionAgent
|
|
|
24
36
|
|
|
25
37
|
private
|
|
26
38
|
|
|
39
|
+
def invalid_authenticity_token
|
|
40
|
+
render json: { error: "Refresh the dashboard and try again", code: "invalid_csrf_token" }, status: :unprocessable_entity
|
|
41
|
+
end
|
|
42
|
+
|
|
27
43
|
# API keys and provider credentials are encrypted at rest, which needs
|
|
28
44
|
# Active Record Encryption keys. The engine derives fallback keys when
|
|
29
45
|
# the host set none (see Engine's action_agent.active_record_encryption
|
|
@@ -3,8 +3,6 @@
|
|
|
3
3
|
module ActionAgent
|
|
4
4
|
module Api
|
|
5
5
|
class DashboardAssistantController < BaseController
|
|
6
|
-
protect_from_forgery with: :exception
|
|
7
|
-
|
|
8
6
|
before_action :require_assistant_enabled!
|
|
9
7
|
before_action :require_owner!
|
|
10
8
|
before_action :require_execution_enabled!, only: :create
|
|
@@ -13,15 +11,6 @@ module ActionAgent
|
|
|
13
11
|
rescue_from DashboardAssistantService::InvalidInput, with: :invalid_input
|
|
14
12
|
rescue_from DashboardAssistantService::ProcessingConsentRequired, with: :processing_consent_required
|
|
15
13
|
rescue_from DashboardAssistantService::SetupRequired, with: :setup_required
|
|
16
|
-
# Rails 8.2 verifies forgery protection from the browser's Sec-Fetch-Site
|
|
17
|
-
# header, renamed the failure to InvalidCrossOriginRequest, and deprecated
|
|
18
|
-
# the old name. Rescue whichever names the running Rails defines, so a
|
|
19
|
-
# rejected request answers with the dashboard's JSON either way.
|
|
20
|
-
# const_defined? does not fire the deprecation the bare constant would.
|
|
21
|
-
rescue_from ActionController::InvalidCrossOriginRequest, with: :invalid_authenticity_token
|
|
22
|
-
if ActionController.const_defined?(:InvalidAuthenticityToken, false)
|
|
23
|
-
rescue_from ActionController::InvalidAuthenticityToken, with: :invalid_authenticity_token
|
|
24
|
-
end
|
|
25
14
|
|
|
26
15
|
def show
|
|
27
16
|
render json: DashboardAssistantService.new(owner: current_owner).configuration
|
|
@@ -0,0 +1,134 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module ActionAgent
|
|
4
|
+
module Api
|
|
5
|
+
# Collector for evaluation reports an application ran itself and
|
|
6
|
+
# published with ActiveAgent::Evals::Publisher:
|
|
7
|
+
# POST <mount>/api/evaluation_reports (e.g. /activeagents/api/evaluation_reports).
|
|
8
|
+
#
|
|
9
|
+
# Authenticated exactly as trace ingest is (IngestAuthentication), and
|
|
10
|
+
# stored by EvaluationReportImport. Responds:
|
|
11
|
+
#
|
|
12
|
+
# 201 — stored; the receipt the publisher checks
|
|
13
|
+
# 200 — an identical retry; the receipt names the stored run. Never
|
|
14
|
+
# refused by the quota or the rate limit.
|
|
15
|
+
# 409 — this run_id already holds a different report
|
|
16
|
+
# 422 — not a valid version-1 report, or an evaluation name the agent
|
|
17
|
+
# already holds for another suite or scope
|
|
18
|
+
# 403 — storing it needs an operator first: a cap on observed agents,
|
|
19
|
+
# evaluations or scenarios, or no owner for the tenant
|
|
20
|
+
# 429 — a new report over the host's quota (kind :evaluation_report) or
|
|
21
|
+
# over RATE_LIMIT new reports a minute from one key
|
|
22
|
+
# 413 — a body over EvaluationReportImport::MAX_BYTES
|
|
23
|
+
# 415 — a body that is not declared application/json
|
|
24
|
+
# 400 — a body that is not JSON
|
|
25
|
+
# 401 — a missing or unknown key
|
|
26
|
+
# 501 — an install with no evaluation tables, or not yet migrated
|
|
27
|
+
# 503 — another import for the same agent held the lock too long
|
|
28
|
+
class EvaluationReportsController < ActionController::API
|
|
29
|
+
include IngestAuthentication
|
|
30
|
+
|
|
31
|
+
# New reports one key may store per minute.
|
|
32
|
+
RATE_LIMIT = 30
|
|
33
|
+
|
|
34
|
+
wrap_parameters false
|
|
35
|
+
|
|
36
|
+
before_action :require_json!
|
|
37
|
+
before_action :require_report_store!
|
|
38
|
+
|
|
39
|
+
# POST <mount>/api/evaluation_reports
|
|
40
|
+
def create
|
|
41
|
+
# The header, not request.content_length, which reads a chunked body
|
|
42
|
+
# in full to measure it.
|
|
43
|
+
return report_too_large if request.get_header("CONTENT_LENGTH").to_i > EvaluationReportImport::MAX_BYTES
|
|
44
|
+
|
|
45
|
+
body = request.body.read(EvaluationReportImport::MAX_BYTES + 1).to_s
|
|
46
|
+
return report_too_large if body.bytesize > EvaluationReportImport::MAX_BYTES
|
|
47
|
+
|
|
48
|
+
run, duplicate = EvaluationReportImport.call(account: @account, payload: JSON.parse(body), admit: -> { admission_denial })
|
|
49
|
+
ActionAgent.record_usage(@account, :evaluation_report) unless duplicate
|
|
50
|
+
|
|
51
|
+
render json: receipt(run, duplicate), status: duplicate ? :ok : :created
|
|
52
|
+
rescue JSON::ParserError
|
|
53
|
+
render json: { error: "Invalid JSON" }, status: :bad_request
|
|
54
|
+
rescue EvaluationReportImport::Invalid, ActiveRecord::RecordInvalid => e
|
|
55
|
+
render json: { error: e.message }, status: :unprocessable_entity
|
|
56
|
+
rescue EvaluationReportImport::Conflict => e
|
|
57
|
+
render json: { error: e.message }, status: :conflict
|
|
58
|
+
rescue EvaluationReportImport::Refused => e
|
|
59
|
+
render json: { error: e.message }, status: :forbidden
|
|
60
|
+
rescue EvaluationReportImport::Denied => e
|
|
61
|
+
render json: e.denial, status: :too_many_requests
|
|
62
|
+
rescue EvaluationReportImport::Busy => e
|
|
63
|
+
response.headers["Retry-After"] = "5"
|
|
64
|
+
render json: { error: e.message }, status: :service_unavailable
|
|
65
|
+
end
|
|
66
|
+
|
|
67
|
+
private
|
|
68
|
+
|
|
69
|
+
# Rails reads and parses a JSON body into params before any callback
|
|
70
|
+
# runs, for the request log among others. With none to parse, the body
|
|
71
|
+
# is read only by #create, and only up to the size limit.
|
|
72
|
+
def process_action(*)
|
|
73
|
+
request.request_parameters = {}
|
|
74
|
+
super
|
|
75
|
+
end
|
|
76
|
+
|
|
77
|
+
# A cross-site page can send a text/plain or form POST without a CORS
|
|
78
|
+
# preflight; it cannot send application/json.
|
|
79
|
+
def require_json!
|
|
80
|
+
return if request.media_type == "application/json"
|
|
81
|
+
|
|
82
|
+
render json: { error: "Content-Type must be application/json" }, status: :unsupported_media_type
|
|
83
|
+
end
|
|
84
|
+
|
|
85
|
+
def require_report_store!
|
|
86
|
+
reason = EvaluationReportImport.unavailable_reason
|
|
87
|
+
render json: { error: reason }, status: :not_implemented if reason
|
|
88
|
+
end
|
|
89
|
+
|
|
90
|
+
# What refuses a report that would be stored, or nil: the rate limit,
|
|
91
|
+
# then the host app's quota checker, asked with kind :evaluation_report.
|
|
92
|
+
# Never asked for an identical retry.
|
|
93
|
+
def admission_denial
|
|
94
|
+
return { error: "Too many evaluation reports; retry in a minute" } if rate_limited?
|
|
95
|
+
|
|
96
|
+
denial = ActionAgent.quota_denial(@account, :evaluation_report)
|
|
97
|
+
quota_denial_body(denial, "Evaluation report limit reached") if denial.present?
|
|
98
|
+
end
|
|
99
|
+
|
|
100
|
+
# Counts a new report against its key's bucket, in the store Rails'
|
|
101
|
+
# own rate_limit uses. One bucket per key: the tenant's on a
|
|
102
|
+
# multi-tenant install, the install's own on a single-tenant one.
|
|
103
|
+
def rate_limited?
|
|
104
|
+
bucket = @account ? "account:#{@account.id}" : "install"
|
|
105
|
+
count = self.class.cache_store.increment("rate-limit:#{controller_path}:#{bucket}", 1, expires_in: 1.minute)
|
|
106
|
+
count.present? && count > RATE_LIMIT
|
|
107
|
+
end
|
|
108
|
+
|
|
109
|
+
def receipt(run, duplicate)
|
|
110
|
+
{
|
|
111
|
+
id: run.id,
|
|
112
|
+
evaluation_id: run.evaluation_id,
|
|
113
|
+
run_id: run.external_run_id,
|
|
114
|
+
status: run.status,
|
|
115
|
+
duplicate: duplicate,
|
|
116
|
+
url: run_url(run)
|
|
117
|
+
}
|
|
118
|
+
end
|
|
119
|
+
|
|
120
|
+
# The dashboard page that shows the run, as a path on this host. A host
|
|
121
|
+
# that routes to this controller from outside the engine's mount
|
|
122
|
+
# overrides it.
|
|
123
|
+
def run_url(run)
|
|
124
|
+
"#{request.script_name}/evaluations/#{run.evaluation_id}/runs/#{run.id}"
|
|
125
|
+
end
|
|
126
|
+
|
|
127
|
+
# 413 by number: Rack named it :payload_too_large before 3.1 and
|
|
128
|
+
# :content_too_large since, and the engine supports both.
|
|
129
|
+
def report_too_large
|
|
130
|
+
render json: { error: "Report exceeds #{EvaluationReportImport::MAX_BYTES / 1.megabyte} MiB" }, status: 413
|
|
131
|
+
end
|
|
132
|
+
end
|
|
133
|
+
end
|
|
134
|
+
end
|
|
@@ -38,22 +38,44 @@ module ActionAgent
|
|
|
38
38
|
# 50 client-side hides an agent whose evaluations are not among the account's
|
|
39
39
|
# 50 most recent. The scope is already restricted to the current user's
|
|
40
40
|
# agents, so an id outside it simply returns nothing.
|
|
41
|
+
#
|
|
42
|
+
# Three fields feed the dashboard's model pickers. They describe the
|
|
43
|
+
# credentials of #picker_credentials_owner:
|
|
44
|
+
# - judge_provider: the provider a judge model runs on, null
|
|
45
|
+
# when none has credentials or the lookup
|
|
46
|
+
# raised
|
|
47
|
+
# - judge_provider_error: true when that lookup raised, as it does
|
|
48
|
+
# when a stored key no longer decrypts
|
|
49
|
+
# - model_providers: the providers agent runs have credentials
|
|
50
|
+
# for, leaving out any whose credentials
|
|
51
|
+
# cannot be read
|
|
41
52
|
def index
|
|
42
53
|
scope = evaluations_scope
|
|
43
54
|
scope = scope.where(agent_id: params[:agent_id]) if params[:agent_id].present?
|
|
44
55
|
evaluations = scope.includes(:agent, :evaluation_runs, :scenarios).recent.limit(50)
|
|
56
|
+
owner = picker_credentials_owner
|
|
45
57
|
|
|
46
|
-
render json: {
|
|
58
|
+
render json: {
|
|
59
|
+
evaluations: evaluations.map { |evaluation| serialize(evaluation) },
|
|
60
|
+
**judge_provider_fields(owner),
|
|
61
|
+
model_providers: AgentExecutionService.available_providers(owner)
|
|
62
|
+
}
|
|
47
63
|
end
|
|
48
64
|
|
|
65
|
+
# Runs listed per evaluation on GET /api/evaluations/:id. The rest of
|
|
66
|
+
# the history stays reachable by run id; `run_count` says how long it is.
|
|
67
|
+
RUN_HISTORY_LIMIT = 20
|
|
68
|
+
|
|
49
69
|
# GET /api/evaluations/:id
|
|
50
70
|
def show
|
|
51
71
|
evaluation = evaluations_scope.find(params[:id])
|
|
72
|
+
run_count = evaluation.evaluation_runs.count
|
|
73
|
+
runs = evaluation.evaluation_runs.recent.limit(RUN_HISTORY_LIMIT).to_a
|
|
52
74
|
|
|
53
75
|
render json: {
|
|
54
76
|
evaluation: serialize(evaluation).merge(
|
|
55
77
|
scenarios: evaluation.scenarios.ordered.map(&:as_json_summary),
|
|
56
|
-
runs:
|
|
78
|
+
runs: runs.each_with_index.map { |run, index| serialize_run(run, number: run_count - index) }
|
|
57
79
|
)
|
|
58
80
|
}
|
|
59
81
|
end
|
|
@@ -98,8 +120,12 @@ module ActionAgent
|
|
|
98
120
|
def run
|
|
99
121
|
evaluation = current_evaluation
|
|
100
122
|
run = start_run(evaluation, selection_params)
|
|
123
|
+
evaluation.reload
|
|
101
124
|
|
|
102
|
-
render json: {
|
|
125
|
+
render json: {
|
|
126
|
+
evaluation: serialize(evaluation),
|
|
127
|
+
run: serialize_run(run, number: evaluation.evaluation_runs.count)
|
|
128
|
+
}
|
|
103
129
|
end
|
|
104
130
|
|
|
105
131
|
# GET /api/evaluations/:id/runs/:run_id
|
|
@@ -116,7 +142,8 @@ module ActionAgent
|
|
|
116
142
|
|
|
117
143
|
render json: {
|
|
118
144
|
evaluation: serialize(evaluation),
|
|
119
|
-
run: serialize_run(run
|
|
145
|
+
run: serialize_run(run, number: run_number(evaluation, run))
|
|
146
|
+
.merge(results: results.map(&:as_json_summary), fix_items: safe_fix_items(run))
|
|
120
147
|
}
|
|
121
148
|
end
|
|
122
149
|
|
|
@@ -213,6 +240,25 @@ module ActionAgent
|
|
|
213
240
|
evaluation.evaluation_runs.create!(status: :failed, error_message: e.message, completed_at: Time.current)
|
|
214
241
|
end
|
|
215
242
|
|
|
243
|
+
# Returns whose credentials the index's model picker fields describe.
|
|
244
|
+
# Agent runs and their judge use the evaluated agent's owner's
|
|
245
|
+
# credentials, so a list scoped to one agent reads that agent's owner,
|
|
246
|
+
# and an unscoped list the signed-in owner.
|
|
247
|
+
def picker_credentials_owner
|
|
248
|
+
agent = owner_agents.find_by(id: params[:agent_id]) if params[:agent_id].present?
|
|
249
|
+
agent ? agent.owner : current_owner
|
|
250
|
+
end
|
|
251
|
+
|
|
252
|
+
# Returns the index's judge_provider and judge_provider_error for
|
|
253
|
+
# +owner+. A lookup that raises, from a key that no longer decrypts or a
|
|
254
|
+
# host credentials hook that fails, is logged and reported as an error.
|
|
255
|
+
def judge_provider_fields(owner)
|
|
256
|
+
{ judge_provider: EvaluationRunnerService.judge_provider_for(owner)&.to_s, judge_provider_error: false }
|
|
257
|
+
rescue StandardError => e
|
|
258
|
+
Rails.logger.warn("[Evaluations] judge provider lookup failed: #{e.class}: #{e.message}")
|
|
259
|
+
{ judge_provider: nil, judge_provider_error: true }
|
|
260
|
+
end
|
|
261
|
+
|
|
216
262
|
def evaluations_scope
|
|
217
263
|
Evaluation.joins(:agent).where(agent: owner_agents)
|
|
218
264
|
end
|
|
@@ -328,7 +374,9 @@ module ActionAgent
|
|
|
328
374
|
end
|
|
329
375
|
|
|
330
376
|
def serialize(evaluation)
|
|
331
|
-
|
|
377
|
+
# size reads the preloaded association on index and COUNTs elsewhere.
|
|
378
|
+
run_count = evaluation.evaluation_runs.size
|
|
379
|
+
latest, previous = recent_runs(evaluation, 2)
|
|
332
380
|
|
|
333
381
|
{
|
|
334
382
|
id: evaluation.id,
|
|
@@ -344,22 +392,50 @@ module ActionAgent
|
|
|
344
392
|
scenario_count: evaluation.scenarios.size,
|
|
345
393
|
scenario_groups: evaluation.scenario_suite? ? evaluation.scenario_groups : [],
|
|
346
394
|
created_at: evaluation.created_at.iso8601,
|
|
347
|
-
|
|
395
|
+
run_count: run_count,
|
|
396
|
+
latest_run: latest ? serialize_run(latest, number: run_count) : nil,
|
|
397
|
+
# Just enough of the run before it for the list to show movement
|
|
398
|
+
# ("+3 passed vs #2") without a request per evaluation.
|
|
399
|
+
previous_run: previous ? serialize_run_summary(previous, number: run_count - 1) : nil
|
|
348
400
|
}
|
|
349
401
|
end
|
|
350
402
|
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
|
|
403
|
+
# Newest first. Sorts the preloaded association when index loaded it
|
|
404
|
+
# rather than issuing one ORDER BY query per evaluation.
|
|
405
|
+
def recent_runs(evaluation, limit)
|
|
406
|
+
runs = evaluation.evaluation_runs
|
|
407
|
+
if runs.loaded?
|
|
408
|
+
runs.sort_by { |run| [ run.created_at, run.id ] }.reverse.first(limit)
|
|
409
|
+
else
|
|
410
|
+
runs.recent.limit(limit).to_a
|
|
411
|
+
end
|
|
412
|
+
end
|
|
413
|
+
|
|
414
|
+
# A run's position in its evaluation's history, oldest = 1.
|
|
415
|
+
def run_number(evaluation, run)
|
|
416
|
+
evaluation.evaluation_runs.where("created_at < ? OR (created_at = ? AND id <= ?)", run.created_at, run.created_at, run.id).count
|
|
417
|
+
end
|
|
418
|
+
|
|
419
|
+
# `number` is the run's position in its evaluation's history, oldest =
|
|
420
|
+
# 1, so the dashboard can say "Run #3" and "vs #2".
|
|
421
|
+
def serialize_run(run, number: nil)
|
|
422
|
+
serialize_run_summary(run, number: number).merge(
|
|
355
423
|
scores: run.scores,
|
|
356
424
|
selection: run.selection,
|
|
357
425
|
models: run.models,
|
|
426
|
+
usage: run.usage,
|
|
427
|
+
error_message: run.error_message
|
|
428
|
+
)
|
|
429
|
+
end
|
|
430
|
+
|
|
431
|
+
def serialize_run_summary(run, number: nil)
|
|
432
|
+
{
|
|
433
|
+
id: run.id,
|
|
434
|
+
number: number,
|
|
435
|
+
status: run.status,
|
|
358
436
|
average_score: safe_average_score(run),
|
|
359
437
|
samples_evaluated: run.samples_evaluated,
|
|
360
438
|
samples_passed: run.samples_passed,
|
|
361
|
-
usage: run.usage,
|
|
362
|
-
error_message: run.error_message,
|
|
363
439
|
completed_at: run.completed_at&.iso8601,
|
|
364
440
|
created_at: run.created_at.iso8601
|
|
365
441
|
}
|
|
@@ -127,13 +127,12 @@ module ActionAgent
|
|
|
127
127
|
end
|
|
128
128
|
|
|
129
129
|
# Traces belong to an agent by foreign key once AgentRegistrar attributes
|
|
130
|
-
# them; older rows predate that, so fall back to the
|
|
130
|
+
# them; older rows predate that, so fall back to the agent's identity.
|
|
131
131
|
def traces_for_agent(agent_id)
|
|
132
132
|
agent = owner_agents.find_by(id: agent_id)
|
|
133
133
|
return ActionAgent.trace_model.none unless agent
|
|
134
134
|
|
|
135
|
-
|
|
136
|
-
.or(ActionAgent.trace_model.where(agent_id: nil, agent_class: agent.telemetry_agent_class))
|
|
135
|
+
owned_traces.where(agent_id: agent.id).or(agent.unattributed_telemetry_traces(owned_traces))
|
|
137
136
|
end
|
|
138
137
|
|
|
139
138
|
# The dashboard-wide time window, shared with Traces. Absent means "all".
|
|
@@ -24,8 +24,10 @@ module ActionAgent
|
|
|
24
24
|
# { "type": "http", "url": "https://activeagents.ai/mcp",
|
|
25
25
|
# "headers": { "Authorization": "Bearer aa_..." } }
|
|
26
26
|
class MCPController < BaseController
|
|
27
|
-
# Authenticated by API key rather than by the host app's sessions
|
|
27
|
+
# Authenticated by API key rather than by the host app's sessions, so
|
|
28
|
+
# there is no session cookie for a cross-site request to ride on.
|
|
28
29
|
allow_unauthenticated_access
|
|
30
|
+
skip_forgery_protection
|
|
29
31
|
before_action :authenticate_api_key!, except: [ :unsupported ]
|
|
30
32
|
|
|
31
33
|
PROTOCOL_VERSION = "2025-03-26"
|
|
@@ -2,14 +2,22 @@
|
|
|
2
2
|
|
|
3
3
|
module ActionAgent
|
|
4
4
|
module Api
|
|
5
|
-
# Model catalogs for the agent builder
|
|
5
|
+
# Model catalogs for the dashboard's model pickers: the agent builder and
|
|
6
|
+
# editor, and the evaluation forms.
|
|
6
7
|
#
|
|
7
8
|
# Hosted providers get a curated list of current models (kept here, server
|
|
8
9
|
# side, so the UI can't drift stale). Ollama is queried live from the
|
|
9
10
|
# account's configured host (Settings -> Provider API Keys, falling back to
|
|
10
11
|
# the platform config) so locally pulled models appear; OpenRouter is
|
|
11
|
-
# queried from its public catalog
|
|
12
|
-
# list on any
|
|
12
|
+
# queried from its public catalog, and Anthropic from its Models API with
|
|
13
|
+
# the account's key. Live lookups fall back to the curated list on any
|
|
14
|
+
# failure.
|
|
15
|
+
#
|
|
16
|
+
# When the host app loads RubyLLM, the chat models its registry lists for
|
|
17
|
+
# the provider that take and return text follow: RubyLLM's bundled
|
|
18
|
+
# catalog, or the host's own model table when it configured one. The list
|
|
19
|
+
# stays de-duplicated, and its first id, the builder's preselected
|
|
20
|
+
# default, is always the live or curated one.
|
|
13
21
|
class ProviderModelsController < BaseController
|
|
14
22
|
before_action :require_owner!
|
|
15
23
|
|
|
@@ -42,7 +50,7 @@ module ActionAgent
|
|
|
42
50
|
source = "curated"
|
|
43
51
|
end
|
|
44
52
|
|
|
45
|
-
render json: { provider: provider, models: models, source: source }
|
|
53
|
+
render json: { provider: provider, models: (models + registry_models(provider)).uniq, source: source }
|
|
46
54
|
end
|
|
47
55
|
|
|
48
56
|
private
|
|
@@ -92,17 +100,43 @@ module ActionAgent
|
|
|
92
100
|
nil
|
|
93
101
|
end
|
|
94
102
|
|
|
103
|
+
# The whole catalog, never a prefix of it: OpenRouter serves several
|
|
104
|
+
# hundred models, and a cap after sorting left everything late in the
|
|
105
|
+
# alphabet (openai/*, qwen/*, ...) unselectable. The editor filters it.
|
|
95
106
|
def live_openrouter_models
|
|
96
107
|
data = Rails.cache.fetch("provider_models:openrouter", expires_in: 1.hour) do
|
|
97
108
|
fetch_json(URI.parse("https://openrouter.ai/api/v1/models"))
|
|
98
109
|
end
|
|
99
110
|
ids = Array(data&.dig("data")).filter_map { |model| model["id"] }
|
|
100
|
-
[ ids.sort
|
|
111
|
+
[ ids.sort, "live" ] if ids.any?
|
|
101
112
|
rescue StandardError => e
|
|
102
113
|
Rails.logger.warn("[ProviderModels] openrouter lookup failed: #{e.message}")
|
|
103
114
|
nil
|
|
104
115
|
end
|
|
105
116
|
|
|
117
|
+
# Returns the ids of the chat models RubyLLM's registry lists for
|
|
118
|
+
# +provider+ that take and return text. Empty without RubyLLM, or when
|
|
119
|
+
# the registry raises. This engine's provider names are RubyLLM's own
|
|
120
|
+
# slugs.
|
|
121
|
+
#
|
|
122
|
+
# `chat_models` alone is not enough: RubyLLM counts a model as a chat
|
|
123
|
+
# model whenever its output modalities name no other kind, none at all
|
|
124
|
+
# included. That takes in the speech, transcription, moderation and
|
|
125
|
+
# completion-only models its registry lists with no modalities, and the
|
|
126
|
+
# transcription models it lists as taking audio. The few chat models it
|
|
127
|
+
# lists with no modalities are left out with them.
|
|
128
|
+
def registry_models(provider)
|
|
129
|
+
return [] unless defined?(::RubyLLM) && ::RubyLLM.respond_to?(:models)
|
|
130
|
+
|
|
131
|
+
::RubyLLM.models.by_provider(provider).chat_models.filter_map do |model|
|
|
132
|
+
modalities = model.modalities
|
|
133
|
+
model.id.presence if Array(modalities&.input).include?("text") && Array(modalities&.output).include?("text")
|
|
134
|
+
end
|
|
135
|
+
rescue StandardError => e
|
|
136
|
+
Rails.logger.warn("[ProviderModels] RubyLLM registry lookup failed: #{e.message}")
|
|
137
|
+
[]
|
|
138
|
+
end
|
|
139
|
+
|
|
106
140
|
def fetch_json(uri)
|
|
107
141
|
response = Net::HTTP.start(
|
|
108
142
|
uri.host, uri.port,
|
|
@@ -16,9 +16,19 @@ module ActionAgent
|
|
|
16
16
|
DEFAULT_LIMIT = 500
|
|
17
17
|
|
|
18
18
|
# GET /api/traces
|
|
19
|
+
#
|
|
20
|
+
# Params:
|
|
21
|
+
# minutes the window in minutes, DEFAULT_WINDOW_MINUTES when absent
|
|
22
|
+
# agent_id an agent the caller can see. Everything in the response,
|
|
23
|
+
# `agents` and `agent_ids` included, is narrowed to that
|
|
24
|
+
# agent's traces (Agent#telemetry_traces). 404 for an agent
|
|
25
|
+
# the caller cannot see.
|
|
26
|
+
# agent an agent_class, narrowing `traces` only
|
|
27
|
+
# service a service_name, narrowing `traces` only
|
|
28
|
+
# status "error" for failed traces only
|
|
19
29
|
def index
|
|
20
30
|
window = params.fetch(:minutes, DEFAULT_WINDOW_MINUTES).to_i.clamp(1, MAX_WINDOW_MINUTES)
|
|
21
|
-
window_scope = traces_scope.for_date_range(window.minutes.ago, Time.current)
|
|
31
|
+
window_scope = agent_scope(traces_scope).for_date_range(window.minutes.ago, Time.current)
|
|
22
32
|
|
|
23
33
|
scope = window_scope
|
|
24
34
|
scope = scope.for_agent(params[:agent]) if params[:agent].present?
|
|
@@ -54,6 +64,15 @@ module ActionAgent
|
|
|
54
64
|
ActionAgent.trace_model.for_account(current_account)
|
|
55
65
|
end
|
|
56
66
|
|
|
67
|
+
# +scope+ narrowed to the traces of the agent `agent_id` names, looked
|
|
68
|
+
# up among the agents the caller can see.
|
|
69
|
+
def agent_scope(scope)
|
|
70
|
+
agent_id = integer_param(:agent_id)
|
|
71
|
+
return scope unless agent_id
|
|
72
|
+
|
|
73
|
+
owner_agents.find(agent_id).telemetry_traces(scope)
|
|
74
|
+
end
|
|
75
|
+
|
|
57
76
|
# One grouped query. A class can appear under several agent records
|
|
58
77
|
# (same class, different action); the most recently active one wins,
|
|
59
78
|
# since that's what the operator most likely means by "this agent".
|
|
@@ -34,8 +34,8 @@ module ActionAgent
|
|
|
34
34
|
# }
|
|
35
35
|
#
|
|
36
36
|
class TracesController < ActionController::API
|
|
37
|
-
|
|
38
|
-
|
|
37
|
+
include IngestAuthentication
|
|
38
|
+
|
|
39
39
|
before_action :enforce_ingest_quota!
|
|
40
40
|
|
|
41
41
|
# POST <mount>/api/traces (e.g. /activeagents/api/traces)
|
|
@@ -75,64 +75,15 @@ module ActionAgent
|
|
|
75
75
|
|
|
76
76
|
private
|
|
77
77
|
|
|
78
|
-
#
|
|
79
|
-
# Only used in multi-tenant mode.
|
|
80
|
-
def authenticate_api_key!
|
|
81
|
-
token = extract_bearer_token
|
|
82
|
-
|
|
83
|
-
if token.blank?
|
|
84
|
-
render json: { error: "Missing Authorization header" }, status: :unauthorized
|
|
85
|
-
return
|
|
86
|
-
end
|
|
87
|
-
|
|
88
|
-
account_class = ActionAgent.account_class.constantize
|
|
89
|
-
@account = account_class.find_by(telemetry_api_key: token)
|
|
90
|
-
|
|
91
|
-
if @account.nil?
|
|
92
|
-
render json: { error: "Invalid API key" }, status: :unauthorized
|
|
93
|
-
return
|
|
94
|
-
end
|
|
95
|
-
|
|
96
|
-
# Track usage for rate limiting (if the account responds to it)
|
|
97
|
-
@account.increment_telemetry_usage! if @account.respond_to?(:increment_telemetry_usage!)
|
|
98
|
-
end
|
|
99
|
-
|
|
100
|
-
# Requires the configured single-tenant ingest key when one is set.
|
|
101
|
-
# The telemetry reporter and ruby_llm_telemetry both send their
|
|
102
|
-
# api_key as a Bearer header, so remote apps work unchanged.
|
|
103
|
-
def authenticate_ingest_key!
|
|
104
|
-
expected = ActionAgent.ingest_api_key
|
|
105
|
-
return if expected.blank?
|
|
106
|
-
|
|
107
|
-
token = extract_bearer_token
|
|
108
|
-
return if token.present? && ActiveSupport::SecurityUtils.secure_compare(token, expected)
|
|
109
|
-
|
|
110
|
-
render json: { error: "Invalid API key" }, status: :unauthorized
|
|
111
|
-
end
|
|
112
|
-
|
|
113
|
-
# The host app's quota checker, asked with kind :trace_ingest — the
|
|
114
|
-
# counterpart to Api::BaseController#enforce_quota!, which asks with
|
|
115
|
-
# :execution. Denials are 429 here rather than 402: a reporter that is
|
|
116
|
-
# over its ingest allowance should back off, not upgrade mid-flush.
|
|
117
|
-
# Same body shape, so a checker's message or Hash payload reads the
|
|
118
|
-
# same on both.
|
|
78
|
+
# The host app's quota checker, asked with kind :trace_ingest.
|
|
119
79
|
def enforce_ingest_quota!
|
|
120
|
-
|
|
121
|
-
return if denial.blank?
|
|
122
|
-
|
|
123
|
-
body = { error: "Trace ingest limit reached" }
|
|
124
|
-
body = denial.is_a?(Hash) ? body.merge(denial) : body.merge(message: denial)
|
|
125
|
-
|
|
126
|
-
render json: body, status: :too_many_requests
|
|
80
|
+
enforce_ingest_quota_for!(:trace_ingest, "Trace ingest limit reached")
|
|
127
81
|
end
|
|
128
82
|
|
|
129
|
-
#
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
match = auth_header.match(/^Bearer\s+(.+)$/i)
|
|
135
|
-
match[1] if match
|
|
83
|
+
# Counts the request against the tenant, when its account model
|
|
84
|
+
# defines the hook.
|
|
85
|
+
def record_ingest_request
|
|
86
|
+
@account.increment_telemetry_usage! if @account.respond_to?(:increment_telemetry_usage!)
|
|
136
87
|
end
|
|
137
88
|
|
|
138
89
|
# Process traces synchronously for local development.
|