activeagent 1.8.0 → 1.8.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -49,6 +49,37 @@ module ActiveAgent
49
49
  diagnosis&.dig("judge", "suggested_tool")
50
50
  end
51
51
 
52
+ # Where the replay's cost came from, as the caller recorded it in the
53
+ # replay metadata: "reported" (the caller priced it), "estimated"
54
+ # (tokens × a model rate), "no_usage" (nothing was generated, so $0.00)
55
+ # or "unpriced" (nothing to price it from). A cost with no record is
56
+ # the caller's own figure; no cost at all is unpriced.
57
+ def cost_source
58
+ recorded = replay_meta("cost_source").to_s
59
+ return recorded if recorded.present?
60
+
61
+ replay.cost.nil? ? "unpriced" : "reported"
62
+ end
63
+
64
+ def estimated_cost?
65
+ cost_source == "estimated"
66
+ end
67
+
68
+ # The rate an estimated cost was worked out at, `{ "input", "output",
69
+ # "source" }` in $ per million tokens, or nil.
70
+ def cost_rate
71
+ rate = replay_meta("cost_rate")
72
+ rate.is_a?(Hash) ? rate.transform_keys(&:to_s) : nil
73
+ end
74
+
75
+ # What the judge spent on this result — `{ "calls", "input_tokens",
76
+ # "output_tokens", "cost", "model", "by_kind", "source" }` — or nil
77
+ # when no judge was asked about it.
78
+ def judge_usage
79
+ usage = replay_meta("judge_usage")
80
+ usage.is_a?(Hash) ? usage.transform_keys(&:to_s) : nil
81
+ end
82
+
52
83
  def to_h
53
84
  {
54
85
  "scenario_key" => scenario.key,
@@ -70,9 +101,18 @@ module ActiveAgent
70
101
  "fault" => fault,
71
102
  "recommendation" => recommendation,
72
103
  "diagnosis" => diagnosis,
104
+ "judge_usage" => judge_usage,
73
105
  "metadata" => replay.metadata
74
106
  }.compact
75
107
  end
108
+
109
+ private
110
+
111
+ # A caller may key the replay metadata with symbols.
112
+ def replay_meta(key)
113
+ metadata = replay.metadata
114
+ metadata[key] || metadata[key.to_sym]
115
+ end
76
116
  end
77
117
  end
78
118
  end
@@ -51,10 +51,13 @@ module ActiveAgent
51
51
  # @param require_judge_scores [Boolean] fail an otherwise passing result
52
52
  # when a requested task/LLM grade is unavailable, rather than falling
53
53
  # back to rule scores. Does not require a judge for rules-only runs.
54
+ # @param release [Hash, nil] the release of the agent under evaluation,
55
+ # `{ "digest", "revision", "label" }`, carried onto the Report so a
56
+ # published run names the code it scored (see Report.new)
54
57
  def initialize(scenarios:, models:, replay:, criteria: [], judge: nil, judge_task: true, available_tools: {},
55
58
  instructions: nil, agent_name: "The agent", threshold: PASS_THRESHOLD,
56
59
  refine_faults: DEFAULT_REFINE_FAULTS, judge_limit: DEFAULT_JUDGE_LIMIT, on_result: nil,
57
- around_evaluation: nil, require_judge_scores: false, metadata: {})
60
+ around_evaluation: nil, require_judge_scores: false, metadata: {}, release: nil)
58
61
  @scenarios = scenarios
59
62
  @models = models
60
63
  @replay = replay
@@ -71,6 +74,7 @@ module ActiveAgent
71
74
  @around_evaluation = around_evaluation
72
75
  @require_judge_scores = require_judge_scores
73
76
  @metadata = metadata
77
+ @release = release
74
78
  @judge_calls = 0
75
79
  @scorer = Scorer.new(criteria: criteria, judge: judge)
76
80
  end
@@ -83,7 +87,7 @@ module ActiveAgent
83
87
  end
84
88
 
85
89
  Report.new(results: results, models: @models, judge: @judge, instructions: @instructions,
86
- threshold: @threshold, metadata: @metadata)
90
+ threshold: @threshold, metadata: @metadata, release: @release)
87
91
  end
88
92
 
89
93
  # Scores and diagnoses one replay. Public so a caller that has already run
@@ -17,6 +17,7 @@ require_relative "evals/scenario_parser"
17
17
  require_relative "evals/suite"
18
18
  require_relative "evals/model_spec"
19
19
  require_relative "evals/replay"
20
+ require_relative "evals/format"
20
21
  require_relative "evals/scorer"
21
22
  require_relative "evals/diagnosis"
22
23
  require_relative "evals/judge"
@@ -1,3 +1,3 @@
1
1
  module ActiveAgent
2
- VERSION = "1.8.0"
2
+ VERSION = "1.8.1"
3
3
  end
metadata CHANGED
@@ -1,14 +1,14 @@
1
1
  --- !ruby/object:Gem::Specification
2
2
  name: activeagent
3
3
  version: !ruby/object:Gem::Version
4
- version: 1.8.0
4
+ version: 1.8.1
5
5
  platform: ruby
6
6
  authors:
7
7
  - Justin Bowen
8
8
  autorequire:
9
9
  bindir: bin
10
10
  cert_chain: []
11
- date: 2026-09-30 00:00:00.000000000 Z
11
+ date: 2026-10-01 00:00:00.000000000 Z
12
12
  dependencies:
13
13
  - !ruby/object:Gem::Dependency
14
14
  name: actionpack
@@ -443,6 +443,7 @@ files:
443
443
  - lib/active_agent/evals/correlation.rb
444
444
  - lib/active_agent/evals/design_tokens.rb
445
445
  - lib/active_agent/evals/diagnosis.rb
446
+ - lib/active_agent/evals/format.rb
446
447
  - lib/active_agent/evals/judge.rb
447
448
  - lib/active_agent/evals/model_spec.rb
448
449
  - lib/active_agent/evals/publisher.rb