activeagent 1.5.2 → 1.6.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +266 -0
- data/lib/active_agent/base.rb +9 -0
- data/lib/active_agent/concerns/authorization.rb +172 -0
- data/lib/active_agent/concerns/parameterized.rb +40 -3
- data/lib/active_agent/concerns/release.rb +171 -0
- data/lib/active_agent/concerns/tooling.rb +3 -1
- data/lib/active_agent/delegation/runner.rb +17 -0
- data/lib/active_agent/evals/diagnosis.rb +62 -6
- data/lib/active_agent/evals/judge.rb +7 -2
- data/lib/active_agent/evals/model_spec.rb +27 -11
- data/lib/active_agent/evals/runner.rb +1 -1
- data/lib/active_agent/evals/scorer.rb +4 -1
- data/lib/active_agent/generation_job.rb +7 -1
- data/lib/active_agent/schema_tools.rb +173 -6
- data/lib/active_agent/telemetry/configuration.rb +11 -0
- data/lib/active_agent/telemetry/instrumentation.rb +9 -0
- data/lib/active_agent/telemetry/tracer.rb +4 -1
- data/lib/active_agent/telemetry.rb +22 -0
- data/lib/active_agent/version.rb +1 -1
- data/lib/active_agent.rb +3 -0
- data/lib/generators/active_agent/schema_tools/USAGE +22 -0
- data/lib/generators/active_agent/schema_tools/schema_tools_generator.rb +84 -0
- data/lib/generators/active_agent/schema_tools/templates/schema_tools.rb.tt +48 -0
- metadata +7 -2
|
@@ -0,0 +1,171 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "digest"
|
|
4
|
+
require "json"
|
|
5
|
+
|
|
6
|
+
module ActiveAgent
|
|
7
|
+
# A release of an agent is what the model is given: the provider and model,
|
|
8
|
+
# the generation options, the prompt templates on disk, the actions and the
|
|
9
|
+
# tools the class declares. {ClassMethods#release_digest} names that
|
|
10
|
+
# deterministically, so two deploys that ship the same agent share a digest
|
|
11
|
+
# and a change to any of those inputs yields a new one — without anyone
|
|
12
|
+
# bumping a number by hand.
|
|
13
|
+
#
|
|
14
|
+
# The digest is what telemetry stamps on every generation (`agent.version`)
|
|
15
|
+
# and what a dashboard cuts an AgentVersion from on deploy, so a trace, a
|
|
16
|
+
# run and an evaluation can all say which release of the agent produced
|
|
17
|
+
# them. {Release.revision} carries the deploy itself (a git SHA or a release
|
|
18
|
+
# label) alongside, when the host knows it.
|
|
19
|
+
#
|
|
20
|
+
# Class-level and memoized: in development a reload replaces the class, so
|
|
21
|
+
# the next reference recomputes it.
|
|
22
|
+
module Release
|
|
23
|
+
extend ActiveSupport::Concern
|
|
24
|
+
|
|
25
|
+
# Option keys that never belong in a manifest: credentials, and per-call
|
|
26
|
+
# state the class does not own.
|
|
27
|
+
EXCLUDED_OPTION_KEYS = %i[
|
|
28
|
+
api_key access_token secret password token trace_id messages message instructions
|
|
29
|
+
].freeze
|
|
30
|
+
SECRET_KEY_PATTERN = /key|token|secret|password|credential/i
|
|
31
|
+
|
|
32
|
+
# How the deploy identifies itself, when it does. A host sets
|
|
33
|
+
# `ActiveAgent::Release.revision = ENV["GIT_SHA"]` (or a proc) from an
|
|
34
|
+
# initializer; otherwise the conventional deploy variables are read.
|
|
35
|
+
class << self
|
|
36
|
+
attr_writer :revision
|
|
37
|
+
|
|
38
|
+
# @return [String, nil]
|
|
39
|
+
def revision
|
|
40
|
+
value = @revision.respond_to?(:call) ? @revision.call : @revision
|
|
41
|
+
value = value.presence || ENV.values_at("SERVICE_VERSION", "GIT_SHA", "KAMAL_VERSION", "SOURCE_VERSION", "HEROKU_SLUG_COMMIT").find(&:present?)
|
|
42
|
+
value&.to_s
|
|
43
|
+
end
|
|
44
|
+
|
|
45
|
+
# Canonical JSON: sorted keys at every level, so the digest does not
|
|
46
|
+
# depend on the order anything was declared in.
|
|
47
|
+
# @api private
|
|
48
|
+
def canonical(value)
|
|
49
|
+
case value
|
|
50
|
+
when Hash then value.map { |k, v| [ k.to_s, canonical(v) ] }.sort_by(&:first).to_h
|
|
51
|
+
when Array then value.map { |v| canonical(v) }
|
|
52
|
+
when Symbol then value.to_s
|
|
53
|
+
else value
|
|
54
|
+
end
|
|
55
|
+
end
|
|
56
|
+
end
|
|
57
|
+
|
|
58
|
+
class_methods do
|
|
59
|
+
# Everything about this class that shapes a generation, as data.
|
|
60
|
+
#
|
|
61
|
+
# @return [Hash]
|
|
62
|
+
def release_manifest
|
|
63
|
+
@release_manifest ||= Release.canonical(
|
|
64
|
+
agent: name,
|
|
65
|
+
provider: release_provider,
|
|
66
|
+
model: prompt_options&.dig(:model),
|
|
67
|
+
options: release_options,
|
|
68
|
+
actions: release_actions,
|
|
69
|
+
templates: release_templates,
|
|
70
|
+
tools: release_tools,
|
|
71
|
+
delegations: release_delegations
|
|
72
|
+
)
|
|
73
|
+
end
|
|
74
|
+
|
|
75
|
+
# A short, stable identifier for {#release_manifest}: the first twelve
|
|
76
|
+
# hex characters of its SHA-256.
|
|
77
|
+
#
|
|
78
|
+
# @return [String]
|
|
79
|
+
def release_digest
|
|
80
|
+
@release_digest ||= Digest::SHA256.hexdigest(JSON.generate(release_manifest))[0, 12]
|
|
81
|
+
end
|
|
82
|
+
|
|
83
|
+
# Forgets the memoized manifest and digest — for a host that edits
|
|
84
|
+
# templates at runtime, and for tests.
|
|
85
|
+
# @return [void]
|
|
86
|
+
def reset_release!
|
|
87
|
+
@release_manifest = nil
|
|
88
|
+
@release_digest = nil
|
|
89
|
+
end
|
|
90
|
+
|
|
91
|
+
private
|
|
92
|
+
|
|
93
|
+
def release_provider
|
|
94
|
+
provider = respond_to?(:prompt_provider) ? prompt_provider : nil
|
|
95
|
+
(provider || prompt_options&.dig(:service))&.to_s
|
|
96
|
+
end
|
|
97
|
+
|
|
98
|
+
# Generation options minus credentials and per-call state. Nested
|
|
99
|
+
# hashes are walked so a token under `options: { headers: … }` is
|
|
100
|
+
# dropped too.
|
|
101
|
+
def release_options
|
|
102
|
+
strip_secrets((prompt_options || {}).except(*EXCLUDED_OPTION_KEYS, :model, :service))
|
|
103
|
+
end
|
|
104
|
+
|
|
105
|
+
def strip_secrets(value)
|
|
106
|
+
case value
|
|
107
|
+
when Hash
|
|
108
|
+
value.each_with_object({}) do |(key, inner), kept|
|
|
109
|
+
next if EXCLUDED_OPTION_KEYS.include?(key.to_sym) || key.to_s.match?(SECRET_KEY_PATTERN)
|
|
110
|
+
|
|
111
|
+
kept[key] = strip_secrets(inner)
|
|
112
|
+
end
|
|
113
|
+
when Array then value.map { |inner| strip_secrets(inner) }
|
|
114
|
+
else value
|
|
115
|
+
end
|
|
116
|
+
end
|
|
117
|
+
|
|
118
|
+
# The public actions — the prompts a caller can invoke.
|
|
119
|
+
def release_actions
|
|
120
|
+
respond_to?(:action_methods) ? action_methods.to_a.sort : []
|
|
121
|
+
end
|
|
122
|
+
|
|
123
|
+
# Every template file under this agent's view prefixes, keyed by its
|
|
124
|
+
# path relative to the view root, with a digest of its contents. The
|
|
125
|
+
# prefixes mirror View#_prefixes without an action: `app/views/<agent>/`
|
|
126
|
+
# and `app/views/agents/<agent without suffix>/`.
|
|
127
|
+
def release_templates
|
|
128
|
+
return {} if anonymous? || !respond_to?(:view_paths)
|
|
129
|
+
|
|
130
|
+
base = name.underscore
|
|
131
|
+
prefixes = [ base, "agents/#{base.delete_suffix("_agent")}" ]
|
|
132
|
+
roots = Array(view_paths).map { |path| path.respond_to?(:to_path) ? path.to_path : path.to_s }
|
|
133
|
+
|
|
134
|
+
roots.each_with_object({}) do |root, templates|
|
|
135
|
+
prefixes.each do |prefix|
|
|
136
|
+
Dir.glob(File.join(root, prefix, "**", "*")).sort.each do |file|
|
|
137
|
+
next unless File.file?(file)
|
|
138
|
+
|
|
139
|
+
relative = file.delete_prefix("#{root}/")
|
|
140
|
+
templates[relative] = Digest::SHA256.hexdigest(File.binread(file))[0, 12]
|
|
141
|
+
end
|
|
142
|
+
end
|
|
143
|
+
end
|
|
144
|
+
end
|
|
145
|
+
|
|
146
|
+
# Tool definitions the class declares itself (a host convention such as
|
|
147
|
+
# schema-derived rosters), reduced to what identifies them.
|
|
148
|
+
def release_tools
|
|
149
|
+
return [] unless respond_to?(:tool_definitions)
|
|
150
|
+
|
|
151
|
+
Array(tool_definitions).map do |definition|
|
|
152
|
+
next definition.to_s unless definition.respond_to?(:to_h)
|
|
153
|
+
|
|
154
|
+
hash = definition.to_h
|
|
155
|
+
{
|
|
156
|
+
name: (hash[:name] || hash["name"]).to_s,
|
|
157
|
+
description: (hash[:description] || hash["description"]).to_s,
|
|
158
|
+
parameters: hash[:parameters] || hash["parameters"]
|
|
159
|
+
}
|
|
160
|
+
end.sort_by { |tool| tool.is_a?(Hash) ? tool[:name] : tool }
|
|
161
|
+
end
|
|
162
|
+
|
|
163
|
+
# The delegations this class declares, by tool name.
|
|
164
|
+
def release_delegations
|
|
165
|
+
return [] unless respond_to?(:delegations)
|
|
166
|
+
|
|
167
|
+
Array(delegations).map { |tool_name, _definition| tool_name.to_s }.sort
|
|
168
|
+
end
|
|
169
|
+
end
|
|
170
|
+
end
|
|
171
|
+
end
|
|
@@ -16,7 +16,9 @@ module ActiveAgent
|
|
|
16
16
|
# @return [Proc] callback proc that accepts (action_name, *args, **kwargs)
|
|
17
17
|
def tools_function
|
|
18
18
|
proc do |action_name, *args, **kwargs|
|
|
19
|
-
|
|
19
|
+
# Marked as a tool call so a refusal is reported to the model as a
|
|
20
|
+
# result rather than raised through the run (see Authorization).
|
|
21
|
+
with_tool_call { process(action_name, *args, **kwargs) }
|
|
20
22
|
end
|
|
21
23
|
end
|
|
22
24
|
end
|
|
@@ -117,6 +117,7 @@ module ActiveAgent
|
|
|
117
117
|
def generate(arguments)
|
|
118
118
|
agent = definition.resolved_agent_class.new
|
|
119
119
|
agent.params = resolved_params(arguments)
|
|
120
|
+
inherit_actor(agent)
|
|
120
121
|
agent.process(definition.action, **arguments)
|
|
121
122
|
|
|
122
123
|
definition.backend.apply(agent)
|
|
@@ -126,6 +127,22 @@ module ActiveAgent
|
|
|
126
127
|
agent.process_prompt
|
|
127
128
|
end
|
|
128
129
|
|
|
130
|
+
# A delegated generation runs on behalf of whoever the parent runs for.
|
|
131
|
+
# The sub-agent gets the parent's caller before its action runs, so its
|
|
132
|
+
# own before_action callbacks and any scope its tools read through decide
|
|
133
|
+
# against the same person — a parent authorized as one user must not
|
|
134
|
+
# hand its specialists an unattributed run, which a correctly written
|
|
135
|
+
# host scope reads as "no access". Hosts on an older framework, where an
|
|
136
|
+
# agent has no caller to carry, are left as they were.
|
|
137
|
+
#
|
|
138
|
+
# @param agent [ActiveAgent::Base]
|
|
139
|
+
# @return [void]
|
|
140
|
+
def inherit_actor(agent)
|
|
141
|
+
return unless owner.respond_to?(:current_user) && agent.respond_to?(:current_user=)
|
|
142
|
+
|
|
143
|
+
agent.current_user = owner.current_user
|
|
144
|
+
end
|
|
145
|
+
|
|
129
146
|
# A delegated generation is part of its parent's work, so it carries the
|
|
130
147
|
# parent's trace id — otherwise the sub-agent's tokens and latency land
|
|
131
148
|
# in a separate trace and the budget you set has nothing to show for it.
|
|
@@ -12,6 +12,7 @@ module ActiveAgent
|
|
|
12
12
|
# tool_error — a tool the agent called returned an error
|
|
13
13
|
# missing_capability — the agent said no tool covers the task
|
|
14
14
|
# expected_tool_not_called — the scenario expects a tool the agent did not call
|
|
15
|
+
# ungrounded_answer — the answer states specifics no tool call supplied
|
|
15
16
|
# forbidden_content — the answer contains a pattern the scenario forbids
|
|
16
17
|
# missing_content — the answer lacks a pattern the scenario expects
|
|
17
18
|
# low_quality — the answer scored below the threshold
|
|
@@ -21,7 +22,7 @@ module ActiveAgent
|
|
|
21
22
|
# Returns nil for a passing result.
|
|
22
23
|
class Diagnosis
|
|
23
24
|
FAULTS = %w[
|
|
24
|
-
run_error tool_error missing_capability expected_tool_not_called
|
|
25
|
+
run_error tool_error missing_capability expected_tool_not_called ungrounded_answer
|
|
25
26
|
forbidden_content missing_content low_quality judge_unavailable
|
|
26
27
|
].freeze
|
|
27
28
|
|
|
@@ -38,6 +39,19 @@ module ActiveAgent
|
|
|
38
39
|
/\bcan(?:'|no)t (?:be )?(?:done|determined|answered) with (?:the|my) (?:current|available) tools\b/i
|
|
39
40
|
].freeze
|
|
40
41
|
|
|
42
|
+
# Phrasings that state a specific fact — a record id, a date, a count of
|
|
43
|
+
# things — which an agent that called no tool can only have invented.
|
|
44
|
+
# Deliberately narrow: a number inside prose ("here are three options",
|
|
45
|
+
# "within 30 days") is not a claim about data, and a false positive here
|
|
46
|
+
# fails a scenario that may have passed on its merits.
|
|
47
|
+
SPECIFIC_CLAIMS = [
|
|
48
|
+
/#\d+\b/,
|
|
49
|
+
/\b\d{4}-\d{2}-\d{2}\b/,
|
|
50
|
+
/\b(?:you have|there are|there is|we have|I found|found|showing|a total of)\s+(?:\*\*)?\d+\b/i,
|
|
51
|
+
/\b\d+\s+(?:\*\*)?(?:open|overdue|pending|active|closed|resolved|completed|unpaid|outstanding|new|matching|
|
|
52
|
+
records?|results?|rows?|entries|items?|tickets?|orders?|tasks?|issues?|invoices?|customers?|users?|milestones?)\b/ix
|
|
53
|
+
].freeze
|
|
54
|
+
|
|
41
55
|
Result = Struct.new(:fault, :summary, :recommendation, :evidence, keyword_init: true) do
|
|
42
56
|
def to_h
|
|
43
57
|
{
|
|
@@ -76,7 +90,7 @@ module ActiveAgent
|
|
|
76
90
|
end
|
|
77
91
|
|
|
78
92
|
def call
|
|
79
|
-
run_error || tool_error || missing_capability || expected_tool_not_called ||
|
|
93
|
+
run_error || tool_error || missing_capability || expected_tool_not_called || ungrounded_answer ||
|
|
80
94
|
forbidden_content || missing_content || low_quality
|
|
81
95
|
end
|
|
82
96
|
|
|
@@ -178,16 +192,58 @@ module ActiveAgent
|
|
|
178
192
|
"#{agent} answered with #{called_tools.uniq.join(', ')} instead of #{expected.join(', ')}. Sharpen " \
|
|
179
193
|
"both tools' descriptions so the model can tell them apart, or say in the instructions which tool " \
|
|
180
194
|
"answers this kind of task."
|
|
195
|
+
elsif asserts_specifics?
|
|
196
|
+
"#{expected.join(', ')} is available but #{agent.downcase} answered without calling any tool and " \
|
|
197
|
+
"stated specifics it could not have looked up (\"#{claim_excerpt}\"). Treat the answer as invented: " \
|
|
198
|
+
"instruct it to answer this kind of task only from a tool result, and to say so when it has none."
|
|
181
199
|
else
|
|
182
200
|
"#{expected.join(', ')} is available but #{agent.downcase} answered without calling any tool. Tell " \
|
|
183
201
|
"it in the instructions to prefer tool-backed answers for this kind of task, and check the tool's " \
|
|
184
202
|
"description says what it returns."
|
|
185
203
|
end
|
|
186
204
|
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
205
|
+
summary = "Expected #{expected.join(' or ')} to be called; #{agent.downcase} called " \
|
|
206
|
+
"#{called_tools.uniq.presence&.join(', ') || 'nothing'}"
|
|
207
|
+
summary += " and answered with specifics no tool supplied" if called_tools.empty? && asserts_specifics?
|
|
208
|
+
|
|
209
|
+
result("expected_tool_not_called", "#{summary}.", recommendation,
|
|
210
|
+
"expected" => expected, "called" => called_tools, "unavailable" => unavailable,
|
|
211
|
+
"ungrounded" => (called_tools.empty? && asserts_specifics?) || nil, "claim" => (claim_excerpt if called_tools.empty?))
|
|
212
|
+
end
|
|
213
|
+
|
|
214
|
+
# The answer states specifics — a count, an id, a date — that no tool
|
|
215
|
+
# call could have supplied. Reached only when the scenario names no
|
|
216
|
+
# expected tool (expected_tool_not_called reports the same fabrication
|
|
217
|
+
# otherwise) and only for an agent that had tools to call: one with
|
|
218
|
+
# none answers from its instructions by design, and whether that is
|
|
219
|
+
# acceptable is the judge's call, not a mechanical one.
|
|
220
|
+
def ungrounded_answer
|
|
221
|
+
return nil if @available_tools.empty? || called_tools.any?
|
|
222
|
+
return nil unless asserts_specifics?
|
|
223
|
+
|
|
224
|
+
result("ungrounded_answer",
|
|
225
|
+
"#{agent} stated specifics (\"#{claim_excerpt}\") without calling any tool that could have supplied them.",
|
|
226
|
+
"Nothing in the answer came from a tool, so the figures in it are invented. Tell #{agent.downcase} in its " \
|
|
227
|
+
"instructions to answer this kind of task only from a tool result and to say when it has none; if none of " \
|
|
228
|
+
"#{@available_tools.join(', ')} returns this data, add a tool that does.",
|
|
229
|
+
"claim" => claim_excerpt, "tools_available" => @available_tools)
|
|
230
|
+
end
|
|
231
|
+
|
|
232
|
+
def asserts_specifics?
|
|
233
|
+
specific_claim.present?
|
|
234
|
+
end
|
|
235
|
+
|
|
236
|
+
def specific_claim
|
|
237
|
+
return @specific_claim if defined?(@specific_claim)
|
|
238
|
+
|
|
239
|
+
@specific_claim = SPECIFIC_CLAIMS.lazy.filter_map { |pattern| answer.match(pattern) }.first
|
|
240
|
+
end
|
|
241
|
+
|
|
242
|
+
def claim_excerpt
|
|
243
|
+
match = specific_claim
|
|
244
|
+
return nil unless match
|
|
245
|
+
|
|
246
|
+
answer[[ match.begin(0) - 40, 0 ].max, 120].to_s.strip
|
|
191
247
|
end
|
|
192
248
|
|
|
193
249
|
def forbidden_content
|
|
@@ -24,6 +24,11 @@ module ActiveAgent
|
|
|
24
24
|
# @yieldparam instructions [String] the system prompt
|
|
25
25
|
# @yieldparam prompt [String] the user prompt
|
|
26
26
|
# @yieldreturn [String] the completion text
|
|
27
|
+
# How much of a scenario's notes the judge reads. Where a suite's notes
|
|
28
|
+
# are its grading rubric, a "Must not…" clause tends to come last, and a
|
|
29
|
+
# judge that never saw it recommends against it.
|
|
30
|
+
NOTES_LIMIT = 1_500
|
|
31
|
+
|
|
27
32
|
def initialize(label:, &generate)
|
|
28
33
|
raise ArgumentError, "Judge.new needs a block that returns the model's completion" unless generate
|
|
29
34
|
|
|
@@ -64,7 +69,7 @@ module ActiveAgent
|
|
|
64
69
|
---
|
|
65
70
|
#{scenario.prompt}
|
|
66
71
|
---
|
|
67
|
-
#{"Context for the evaluator: #{scenario.notes.truncate(
|
|
72
|
+
#{"Context for the evaluator: #{scenario.notes.truncate(NOTES_LIMIT)}\n" if scenario.notes.present?}
|
|
68
73
|
The assistant answered:
|
|
69
74
|
---
|
|
70
75
|
#{answer.to_s.truncate(4_000)}
|
|
@@ -99,7 +104,7 @@ module ActiveAgent
|
|
|
99
104
|
Scenario (the user's message):
|
|
100
105
|
#{scenario.prompt}
|
|
101
106
|
#{"Expected tools: #{scenario.expected_tools.join(', ')}" if scenario.expected_tools.any?}
|
|
102
|
-
#{"Notes: #{scenario.notes.truncate(
|
|
107
|
+
#{"Notes: #{scenario.notes.truncate(NOTES_LIMIT)}" if scenario.notes.present?}
|
|
103
108
|
|
|
104
109
|
Tools the agent called:
|
|
105
110
|
#{calls.presence || '(none)'}
|
|
@@ -46,22 +46,38 @@ module ActiveAgent
|
|
|
46
46
|
# duplicates by label.
|
|
47
47
|
def self.parse_all(values, **options)
|
|
48
48
|
values = values.to_s.split(",") unless values.is_a?(Array)
|
|
49
|
-
values.
|
|
49
|
+
values.filter_map { |value| from_value(value, **options) }.uniq(&:label)
|
|
50
50
|
end
|
|
51
51
|
|
|
52
|
-
#
|
|
53
|
-
#
|
|
54
|
-
# re-run
|
|
55
|
-
#
|
|
56
|
-
#
|
|
52
|
+
# One requested model, from the text a user typed or from a spec handed
|
|
53
|
+
# back whole. The dashboard persists `specs.map(&:to_h)` and returns it on
|
|
54
|
+
# a re-run, so a value may be a Hash: one that names both `provider` and
|
|
55
|
+
# `model` is rebuilt exactly as it ran, because re-parsing its label
|
|
56
|
+
# would route a vendor-prefixed model the wrong way —
|
|
57
|
+
# `"anthropic/claude-sonnet-4.5"` run through OpenRouter came back as
|
|
58
|
+
# Anthropic's own `claude-sonnet-4.5` the moment that provider was
|
|
59
|
+
# installed. A Hash naming only a label or a model is parsed from that
|
|
60
|
+
# text; anything naming nothing is dropped.
|
|
57
61
|
#
|
|
58
|
-
# @return [
|
|
59
|
-
def self.
|
|
60
|
-
|
|
62
|
+
# @return [ModelSpec, nil]
|
|
63
|
+
def self.from_value(value, **options)
|
|
64
|
+
text =
|
|
65
|
+
if value.respond_to?(:to_h) && !value.is_a?(String)
|
|
66
|
+
hash = value.to_h.stringify_keys
|
|
67
|
+
if hash["provider"].present? && hash["model"].present?
|
|
68
|
+
return new(label: hash["label"].presence || hash["model"], provider: hash["provider"], model: hash["model"])
|
|
69
|
+
end
|
|
61
70
|
|
|
62
|
-
|
|
71
|
+
hash.values_at("label", "model").compact.first.to_s.strip
|
|
72
|
+
else
|
|
73
|
+
value.to_s.strip
|
|
74
|
+
end
|
|
75
|
+
|
|
76
|
+
return nil if text.blank?
|
|
77
|
+
|
|
78
|
+
parse(text, **options)
|
|
63
79
|
end
|
|
64
|
-
private_class_method :
|
|
80
|
+
private_class_method :from_value
|
|
65
81
|
|
|
66
82
|
# The provider a bare model name runs under. A rule whose provider the
|
|
67
83
|
# caller does not offer is skipped, so an app without Ollama does not
|
|
@@ -28,7 +28,7 @@ module ActiveAgent
|
|
|
28
28
|
class Runner
|
|
29
29
|
# Faults where a judge can add something the evidence alone cannot: what
|
|
30
30
|
# tool to add, or how to change the instructions.
|
|
31
|
-
DEFAULT_REFINE_FAULTS = %w[missing_capability expected_tool_not_called low_quality missing_content].freeze
|
|
31
|
+
DEFAULT_REFINE_FAULTS = %w[missing_capability expected_tool_not_called ungrounded_answer low_quality missing_content].freeze
|
|
32
32
|
DEFAULT_JUDGE_LIMIT = 25
|
|
33
33
|
|
|
34
34
|
attr_reader :scenarios, :models, :criteria, :judge, :threshold
|
|
@@ -49,7 +49,10 @@ module ActiveAgent
|
|
|
49
49
|
hit = scenario.forbidden_patterns.any? { |pattern| self.class.matches_pattern?(answer, pattern) }
|
|
50
50
|
scores["forbidden_content"] = hit ? 0.0 : 1.0
|
|
51
51
|
end
|
|
52
|
-
|
|
52
|
+
# A tool that ran without erroring is evidence only when it is one the
|
|
53
|
+
# scenario expected: a wrong tool that succeeded used to outscore
|
|
54
|
+
# calling nothing at all.
|
|
55
|
+
if replay.tool_calls.any? && (scenario.expected_tools.empty? || (scenario.expected_tools & replay.tool_names).any?)
|
|
53
56
|
scores["tools_succeeded"] = replay.failed_tool_calls.any? ? 0.0 : 1.0
|
|
54
57
|
end
|
|
55
58
|
|
|
@@ -18,8 +18,14 @@ module ActiveAgent
|
|
|
18
18
|
|
|
19
19
|
rescue_from StandardError, with: :handle_exception_with_agent_class
|
|
20
20
|
|
|
21
|
-
|
|
21
|
+
# +actor+ is the caller the generation runs on behalf of
|
|
22
|
+
# (ActiveAgent::Authorization). ActiveJob serializes it like any other
|
|
23
|
+
# argument, so a record arrives as the same record the caller passed and
|
|
24
|
+
# the agent's authorization callbacks decide against a real user rather
|
|
25
|
+
# than against nil.
|
|
26
|
+
def perform(agent, agent_method, generation_method, args:, kwargs: nil, params: nil, actor: nil)
|
|
22
27
|
agent_class = params ? agent.constantize.with(params) : agent.constantize
|
|
28
|
+
agent_class = agent_class.as(actor) if actor
|
|
23
29
|
prompt = if kwargs
|
|
24
30
|
agent_class.public_send(agent_method, *args, **kwargs)
|
|
25
31
|
else
|
|
@@ -68,6 +68,21 @@ module ActiveAgent
|
|
|
68
68
|
# ask for 10_000; this is what stops that from becoming the prompt.
|
|
69
69
|
MAX_LIMIT = 100
|
|
70
70
|
|
|
71
|
+
# Column types a range comparison is offered for. Strings and booleans
|
|
72
|
+
# are deliberately absent: a lexical `>` on a name column answers a
|
|
73
|
+
# question nobody asked.
|
|
74
|
+
RANGE_FILTERABLE_TYPES = %i[date datetime time integer float decimal].freeze
|
|
75
|
+
|
|
76
|
+
# The comparison operators a range filter may use, mapped to the Arel
|
|
77
|
+
# predicate that builds them. Names are the ones models reach for
|
|
78
|
+
# unprompted (`before`/`after` for dates, `lt`/`gte` for numbers), so a
|
|
79
|
+
# reasonable guess resolves instead of erroring.
|
|
80
|
+
RANGE_OPERATORS = {
|
|
81
|
+
"before" => :lt, "after" => :gt,
|
|
82
|
+
"lt" => :lt, "lte" => :lteq, "gt" => :gt, "gte" => :gteq,
|
|
83
|
+
"on_or_before" => :lteq, "on_or_after" => :gteq
|
|
84
|
+
}.freeze
|
|
85
|
+
|
|
71
86
|
# Raised when a tool call names a column outside the declared allowlists,
|
|
72
87
|
# or is otherwise outside the declared boundary.
|
|
73
88
|
class UnpermittedAttribute < ArgumentError; end
|
|
@@ -168,6 +183,71 @@ module ActiveAgent
|
|
|
168
183
|
@scope = block
|
|
169
184
|
end
|
|
170
185
|
|
|
186
|
+
# Runtime-defined tool classes, keyed by model name. A definition built
|
|
187
|
+
# by {.define} replaces the previous one for its model, so a registry
|
|
188
|
+
# that is rebuilt on every change — from a table, from a dashboard
|
|
189
|
+
# edit — holds one class per model rather than one per rebuild.
|
|
190
|
+
#
|
|
191
|
+
# @return [Hash{String => Class}]
|
|
192
|
+
def registry
|
|
193
|
+
@registry ||= {}
|
|
194
|
+
end
|
|
195
|
+
|
|
196
|
+
# Builds a tool class from a declaration rather than from a file:
|
|
197
|
+
#
|
|
198
|
+
# ActiveAgent::SchemaTools.define(Reservation,
|
|
199
|
+
# filterable: %i[status guest_id],
|
|
200
|
+
# returns: %i[id status guest_id arrives_on],
|
|
201
|
+
# policy: true) # ReservationPolicy::Scope, as scope_by_policy would
|
|
202
|
+
#
|
|
203
|
+
# The class behaves exactly as a file-defined one — same roster, same
|
|
204
|
+
# allowlists, same +call+ — and is named "<Model>Tools" for logs and
|
|
205
|
+
# telemetry. It is marked runtime-built so discovery does not read it
|
|
206
|
+
# back out of +descendants+ (where every class ever built stays until
|
|
207
|
+
# collected), and registered under its model, replacing whatever the
|
|
208
|
+
# registry held: that is what keeps a rebuild from accumulating classes
|
|
209
|
+
# (#441). +scope:+ takes a lambda or proc receiving the actor; +policy:+
|
|
210
|
+
# resolves the model's policy by name; neither means unscoped, as for a
|
|
211
|
+
# file-defined class.
|
|
212
|
+
#
|
|
213
|
+
# @param model [Class] an ActiveRecord class
|
|
214
|
+
# @param filterable [Array<Symbol, String>]
|
|
215
|
+
# @param returns [Array<Symbol, String>]
|
|
216
|
+
# @param scope [Proc, nil]
|
|
217
|
+
# @param policy [Boolean, Class] true for the conventional policy, or the policy class
|
|
218
|
+
# @param name [String, nil] the class name, "<Model>Tools" by default
|
|
219
|
+
# @return [Class]
|
|
220
|
+
def define(model, filterable: [], returns: [], scope: nil, policy: false, name: nil)
|
|
221
|
+
klass = Class.new(self)
|
|
222
|
+
klass.instance_variable_set(:@runtime, true)
|
|
223
|
+
class_name = name || "#{model.name}Tools"
|
|
224
|
+
klass.define_singleton_method(:name) { class_name }
|
|
225
|
+
klass.model(model)
|
|
226
|
+
klass.filterable(*filterable) if filterable.present?
|
|
227
|
+
klass.returns(*returns) if returns.present?
|
|
228
|
+
klass.scope_by_policy(policy == true ? nil : policy) if policy
|
|
229
|
+
klass.scope(&scope) if scope
|
|
230
|
+
|
|
231
|
+
registry[model.name] = klass
|
|
232
|
+
end
|
|
233
|
+
|
|
234
|
+
# Drops the runtime definition for +model+; discovery stops offering it.
|
|
235
|
+
#
|
|
236
|
+
# @param model [Class, String]
|
|
237
|
+
# @return [Class, nil] the class that was registered
|
|
238
|
+
def undefine(model)
|
|
239
|
+
registry.delete(model.respond_to?(:name) ? model.name : model.to_s)
|
|
240
|
+
end
|
|
241
|
+
|
|
242
|
+
# Whether this class was built by {.define} rather than loaded from a
|
|
243
|
+
# file. Discovery reads runtime classes from {.registry}, never from
|
|
244
|
+
# +descendants+, so a superseded one is not offered twice.
|
|
245
|
+
#
|
|
246
|
+
# @return [Boolean]
|
|
247
|
+
def runtime?
|
|
248
|
+
@runtime == true
|
|
249
|
+
end
|
|
250
|
+
|
|
171
251
|
# The full, fixed tool roster in provider function-calling format.
|
|
172
252
|
#
|
|
173
253
|
# @return [Array<Hash>] tool definitions with :name, :description, :parameters
|
|
@@ -235,8 +315,13 @@ module ActiveAgent
|
|
|
235
315
|
|
|
236
316
|
# Validates and normalizes a filter hash against the allowlist.
|
|
237
317
|
#
|
|
318
|
+
# A filter value is normally matched for equality. A Hash value instead
|
|
319
|
+
# declares a range — `{ "before" => "2026-01-01" }`, `{ "gte" => 10 }` —
|
|
320
|
+
# and may carry two bounds at once to express a window.
|
|
321
|
+
#
|
|
238
322
|
# @api private
|
|
239
|
-
# @raise [UnpermittedAttribute] if any key is not declared filterable
|
|
323
|
+
# @raise [UnpermittedAttribute] if any key is not declared filterable,
|
|
324
|
+
# or a range names an operator that does not exist
|
|
240
325
|
def permitted_filters!(arguments)
|
|
241
326
|
filters = arguments.each_with_object({}) do |(key, value), memo|
|
|
242
327
|
next if value.nil?
|
|
@@ -253,6 +338,54 @@ module ActiveAgent
|
|
|
253
338
|
filters
|
|
254
339
|
end
|
|
255
340
|
|
|
341
|
+
# Splits filters into equality pairs and range predicates.
|
|
342
|
+
#
|
|
343
|
+
# Kept separate from {.permitted_filters!} because the two halves are
|
|
344
|
+
# applied differently: equality goes to `where(hash)`, ranges have to be
|
|
345
|
+
# built through Arel.
|
|
346
|
+
#
|
|
347
|
+
# @api private
|
|
348
|
+
# @return [Array(Hash, Array<Arel::Nodes::Node>)]
|
|
349
|
+
def partition_filters!(filters)
|
|
350
|
+
equality = {}
|
|
351
|
+
ranges = []
|
|
352
|
+
|
|
353
|
+
filters.each do |column, value|
|
|
354
|
+
if value.is_a?(Hash)
|
|
355
|
+
ranges.concat(range_predicates!(column, value))
|
|
356
|
+
else
|
|
357
|
+
equality[column] = value
|
|
358
|
+
end
|
|
359
|
+
end
|
|
360
|
+
|
|
361
|
+
[ equality, ranges ]
|
|
362
|
+
end
|
|
363
|
+
|
|
364
|
+
# Builds Arel predicates for one column's range hash.
|
|
365
|
+
#
|
|
366
|
+
# Rails silently turns `where(col: { "before" => x })` into `col = NULL`,
|
|
367
|
+
# which matches nothing and reports zero rather than failing — the worst
|
|
368
|
+
# outcome for an agent, which reads it as a truthful empty answer. So an
|
|
369
|
+
# unknown operator is rejected loudly here instead.
|
|
370
|
+
#
|
|
371
|
+
# @api private
|
|
372
|
+
# @raise [UnpermittedAttribute] on an unknown operator
|
|
373
|
+
def range_predicates!(column, value)
|
|
374
|
+
arel = @model.arel_table[column]
|
|
375
|
+
type = @model.type_for_attribute(column)
|
|
376
|
+
|
|
377
|
+
value.map do |operator, operand|
|
|
378
|
+
predicate = RANGE_OPERATORS[operator.to_s]
|
|
379
|
+
unless predicate
|
|
380
|
+
raise UnpermittedAttribute,
|
|
381
|
+
"`#{operator}` is not a valid comparison for `#{column}`. " \
|
|
382
|
+
"Allowed comparisons: #{RANGE_OPERATORS.keys.join(", ")}"
|
|
383
|
+
end
|
|
384
|
+
|
|
385
|
+
arel.public_send(predicate, type.cast(operand))
|
|
386
|
+
end
|
|
387
|
+
end
|
|
388
|
+
|
|
256
389
|
# Projects a record down to the declared return columns.
|
|
257
390
|
#
|
|
258
391
|
# The projection happens in SQL (+select+) as well as here, but the Ruby
|
|
@@ -321,7 +454,40 @@ module ActiveAgent
|
|
|
321
454
|
)
|
|
322
455
|
properties = schema[:schema][:properties]
|
|
323
456
|
|
|
324
|
-
filterable.index_with
|
|
457
|
+
filterable.index_with do |column|
|
|
458
|
+
scalar = (properties[column] || { type: "string" }).deep_dup
|
|
459
|
+
range_filterable?(column) ? with_range_form(column, scalar) : scalar
|
|
460
|
+
end
|
|
461
|
+
end
|
|
462
|
+
|
|
463
|
+
# Dates, times and numbers are the columns a question like "overdue" or
|
|
464
|
+
# "more than 10" actually needs a comparison on.
|
|
465
|
+
def range_filterable?(column)
|
|
466
|
+
RANGE_FILTERABLE_TYPES.include?(@model.type_for_attribute(column).type)
|
|
467
|
+
end
|
|
468
|
+
|
|
469
|
+
# Offers a column as either a scalar (equality) or a range object.
|
|
470
|
+
#
|
|
471
|
+
# Without this the range form works but is undiscoverable: a model shown
|
|
472
|
+
# only `{type: "string", format: "date"}` has no way to know it may ask
|
|
473
|
+
# for `before`, and answers date questions with an equality match or no
|
|
474
|
+
# filter at all.
|
|
475
|
+
def with_range_form(column, scalar)
|
|
476
|
+
operand = scalar.slice(:type, :format)
|
|
477
|
+
description = scalar[:description]
|
|
478
|
+
|
|
479
|
+
{
|
|
480
|
+
description: [ description, "Accepts an exact value, or a range object such as " \
|
|
481
|
+
"{\"before\": ...} / {\"gte\": ...} (#{RANGE_OPERATORS.keys.join(", ")})." ].compact.join(" "),
|
|
482
|
+
anyOf: [
|
|
483
|
+
scalar.except(:description),
|
|
484
|
+
{
|
|
485
|
+
type: "object",
|
|
486
|
+
properties: RANGE_OPERATORS.keys.index_with { operand.dup },
|
|
487
|
+
additionalProperties: false
|
|
488
|
+
}
|
|
489
|
+
]
|
|
490
|
+
}
|
|
325
491
|
end
|
|
326
492
|
|
|
327
493
|
def resource_name
|
|
@@ -350,10 +516,10 @@ module ActiveAgent
|
|
|
350
516
|
)
|
|
351
517
|
|
|
352
518
|
define_singleton_method(name) do |actor: nil, limit: nil, **arguments|
|
|
353
|
-
|
|
519
|
+
equality, ranges = partition_filters!(permitted_filters!(arguments))
|
|
354
520
|
capped = normalize_limit(limit)
|
|
355
521
|
|
|
356
|
-
relation = relation_for(actor).where(
|
|
522
|
+
relation = ranges.reduce(relation_for(actor).where(equality)) { |rel, p| rel.where(p) }
|
|
357
523
|
# One extra row distinguishes "exactly at the limit" from "more than
|
|
358
524
|
# the limit", without a second COUNT query.
|
|
359
525
|
records = relation.limit(capped + 1).to_a
|
|
@@ -378,9 +544,10 @@ module ActiveAgent
|
|
|
378
544
|
)
|
|
379
545
|
|
|
380
546
|
define_singleton_method(name) do |actor: nil, **arguments|
|
|
381
|
-
|
|
547
|
+
equality, ranges = partition_filters!(permitted_filters!(arguments))
|
|
548
|
+
relation = ranges.reduce(relation_for(actor).where(equality)) { |rel, p| rel.where(p) }
|
|
382
549
|
|
|
383
|
-
{ count:
|
|
550
|
+
{ count: relation.count }
|
|
384
551
|
end
|
|
385
552
|
end
|
|
386
553
|
|