activeagent 1.8.0 → 1.8.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +149 -0
- data/lib/active_agent/evals/diagnosis.rb +3 -3
- data/lib/active_agent/evals/format.rb +115 -0
- data/lib/active_agent/evals/judge.rb +2 -0
- data/lib/active_agent/evals/report.rb +232 -31
- data/lib/active_agent/evals/report_html.rb +316 -149
- data/lib/active_agent/evals/result.rb +40 -0
- data/lib/active_agent/evals/runner.rb +6 -2
- data/lib/active_agent/evals.rb +1 -0
- data/lib/active_agent/version.rb +1 -1
- metadata +3 -2
|
@@ -49,6 +49,37 @@ module ActiveAgent
|
|
|
49
49
|
diagnosis&.dig("judge", "suggested_tool")
|
|
50
50
|
end
|
|
51
51
|
|
|
52
|
+
# Where the replay's cost came from, as the caller recorded it in the
|
|
53
|
+
# replay metadata: "reported" (the caller priced it), "estimated"
|
|
54
|
+
# (tokens × a model rate), "no_usage" (nothing was generated, so $0.00)
|
|
55
|
+
# or "unpriced" (nothing to price it from). A cost with no record is
|
|
56
|
+
# the caller's own figure; no cost at all is unpriced.
|
|
57
|
+
def cost_source
|
|
58
|
+
recorded = replay_meta("cost_source").to_s
|
|
59
|
+
return recorded if recorded.present?
|
|
60
|
+
|
|
61
|
+
replay.cost.nil? ? "unpriced" : "reported"
|
|
62
|
+
end
|
|
63
|
+
|
|
64
|
+
def estimated_cost?
|
|
65
|
+
cost_source == "estimated"
|
|
66
|
+
end
|
|
67
|
+
|
|
68
|
+
# The rate an estimated cost was worked out at, `{ "input", "output",
|
|
69
|
+
# "source" }` in $ per million tokens, or nil.
|
|
70
|
+
def cost_rate
|
|
71
|
+
rate = replay_meta("cost_rate")
|
|
72
|
+
rate.is_a?(Hash) ? rate.transform_keys(&:to_s) : nil
|
|
73
|
+
end
|
|
74
|
+
|
|
75
|
+
# What the judge spent on this result — `{ "calls", "input_tokens",
|
|
76
|
+
# "output_tokens", "cost", "model", "by_kind", "source" }` — or nil
|
|
77
|
+
# when no judge was asked about it.
|
|
78
|
+
def judge_usage
|
|
79
|
+
usage = replay_meta("judge_usage")
|
|
80
|
+
usage.is_a?(Hash) ? usage.transform_keys(&:to_s) : nil
|
|
81
|
+
end
|
|
82
|
+
|
|
52
83
|
def to_h
|
|
53
84
|
{
|
|
54
85
|
"scenario_key" => scenario.key,
|
|
@@ -70,9 +101,18 @@ module ActiveAgent
|
|
|
70
101
|
"fault" => fault,
|
|
71
102
|
"recommendation" => recommendation,
|
|
72
103
|
"diagnosis" => diagnosis,
|
|
104
|
+
"judge_usage" => judge_usage,
|
|
73
105
|
"metadata" => replay.metadata
|
|
74
106
|
}.compact
|
|
75
107
|
end
|
|
108
|
+
|
|
109
|
+
private
|
|
110
|
+
|
|
111
|
+
# A caller may key the replay metadata with symbols.
|
|
112
|
+
def replay_meta(key)
|
|
113
|
+
metadata = replay.metadata
|
|
114
|
+
metadata[key] || metadata[key.to_sym]
|
|
115
|
+
end
|
|
76
116
|
end
|
|
77
117
|
end
|
|
78
118
|
end
|
|
@@ -51,10 +51,13 @@ module ActiveAgent
|
|
|
51
51
|
# @param require_judge_scores [Boolean] fail an otherwise passing result
|
|
52
52
|
# when a requested task/LLM grade is unavailable, rather than falling
|
|
53
53
|
# back to rule scores. Does not require a judge for rules-only runs.
|
|
54
|
+
# @param release [Hash, nil] the release of the agent under evaluation,
|
|
55
|
+
# `{ "digest", "revision", "label" }`, carried onto the Report so a
|
|
56
|
+
# published run names the code it scored (see Report.new)
|
|
54
57
|
def initialize(scenarios:, models:, replay:, criteria: [], judge: nil, judge_task: true, available_tools: {},
|
|
55
58
|
instructions: nil, agent_name: "The agent", threshold: PASS_THRESHOLD,
|
|
56
59
|
refine_faults: DEFAULT_REFINE_FAULTS, judge_limit: DEFAULT_JUDGE_LIMIT, on_result: nil,
|
|
57
|
-
around_evaluation: nil, require_judge_scores: false, metadata: {})
|
|
60
|
+
around_evaluation: nil, require_judge_scores: false, metadata: {}, release: nil)
|
|
58
61
|
@scenarios = scenarios
|
|
59
62
|
@models = models
|
|
60
63
|
@replay = replay
|
|
@@ -71,6 +74,7 @@ module ActiveAgent
|
|
|
71
74
|
@around_evaluation = around_evaluation
|
|
72
75
|
@require_judge_scores = require_judge_scores
|
|
73
76
|
@metadata = metadata
|
|
77
|
+
@release = release
|
|
74
78
|
@judge_calls = 0
|
|
75
79
|
@scorer = Scorer.new(criteria: criteria, judge: judge)
|
|
76
80
|
end
|
|
@@ -83,7 +87,7 @@ module ActiveAgent
|
|
|
83
87
|
end
|
|
84
88
|
|
|
85
89
|
Report.new(results: results, models: @models, judge: @judge, instructions: @instructions,
|
|
86
|
-
threshold: @threshold, metadata: @metadata)
|
|
90
|
+
threshold: @threshold, metadata: @metadata, release: @release)
|
|
87
91
|
end
|
|
88
92
|
|
|
89
93
|
# Scores and diagnoses one replay. Public so a caller that has already run
|
data/lib/active_agent/evals.rb
CHANGED
|
@@ -17,6 +17,7 @@ require_relative "evals/scenario_parser"
|
|
|
17
17
|
require_relative "evals/suite"
|
|
18
18
|
require_relative "evals/model_spec"
|
|
19
19
|
require_relative "evals/replay"
|
|
20
|
+
require_relative "evals/format"
|
|
20
21
|
require_relative "evals/scorer"
|
|
21
22
|
require_relative "evals/diagnosis"
|
|
22
23
|
require_relative "evals/judge"
|
data/lib/active_agent/version.rb
CHANGED
metadata
CHANGED
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
--- !ruby/object:Gem::Specification
|
|
2
2
|
name: activeagent
|
|
3
3
|
version: !ruby/object:Gem::Version
|
|
4
|
-
version: 1.8.
|
|
4
|
+
version: 1.8.1
|
|
5
5
|
platform: ruby
|
|
6
6
|
authors:
|
|
7
7
|
- Justin Bowen
|
|
8
8
|
autorequire:
|
|
9
9
|
bindir: bin
|
|
10
10
|
cert_chain: []
|
|
11
|
-
date: 2026-
|
|
11
|
+
date: 2026-10-01 00:00:00.000000000 Z
|
|
12
12
|
dependencies:
|
|
13
13
|
- !ruby/object:Gem::Dependency
|
|
14
14
|
name: actionpack
|
|
@@ -443,6 +443,7 @@ files:
|
|
|
443
443
|
- lib/active_agent/evals/correlation.rb
|
|
444
444
|
- lib/active_agent/evals/design_tokens.rb
|
|
445
445
|
- lib/active_agent/evals/diagnosis.rb
|
|
446
|
+
- lib/active_agent/evals/format.rb
|
|
446
447
|
- lib/active_agent/evals/judge.rb
|
|
447
448
|
- lib/active_agent/evals/model_spec.rb
|
|
448
449
|
- lib/active_agent/evals/publisher.rb
|