turnkit 0.8.0 → 0.9.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
checksums.yaml CHANGED
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  SHA256:
3
- metadata.gz: ceb5e53317935ed6bb840b8b6e06157138fb95027f35ed798259a7c9bebbbcbd
4
- data.tar.gz: e4c9f08152c55fa2e7619285f723fb1cba673a9760e9952b8d6792f87333ae71
3
+ metadata.gz: 73e85ee51e846b0d62828d964ad53b00fd0f6ce1e11d56611b47b45462d49d9f
4
+ data.tar.gz: f4c4591902e4a9a9fc4c2a50c6e768ad6a4191ac242418434ff5d8b22e99143b
5
5
  SHA512:
6
- metadata.gz: eb629f09b9566f782014dbefaaf98a2e238c9cd39728628360a09147f117d44a043c513f8191ad1f89cd9b63c428b52fef27679d28d3a02737d64bff8762b1cc
7
- data.tar.gz: 290d3a6c992c30deac5923f7c74a86f25b5fcf8fef851a84ccceae4f72e9fe3f32d9768efff08b465c2c2b4144140306b2182c9ef4168aa2f54e8555805206d7
6
+ metadata.gz: cd39ba14aa1e48ce133d8b4cffc76d63939494a90466cfd62a5b70cb612f76f2caf8a5ffafc371272f00644ff1842a9b96768fea0724788721c45c3d36163655
7
+ data.tar.gz: a77ae5606a43be2790b54cf2a6c8eb78225b5f8b41d0f7b166b8c67b6568a357e65d1ae8f690b446c55afde370aab33d1d3f21dad30a7aec493eb2eb6007224c
data/CHANGELOG.md CHANGED
@@ -1,5 +1,15 @@
1
1
  # Changelog
2
2
 
3
+ ## 0.9.0 - 2026-09-19
4
+
5
+ - Add `Turn#internal_evaluation` and a dependency-free Cloudflare Jev adapter for
6
+ Noul/Choice/Score judgments, separate from chat. Persist request/attempt receipts
7
+ with fenced atomic usage/cost accounting and bounded opt-in retries.
8
+ - Add optional runtime-owned `output_metadata` to child results so output-audit
9
+ callables can report assessments without changing strict generated schemas.
10
+ - Preserve unknown evaluation costs in cost summaries and retain total-only
11
+ provider charges when aggregating them with itemized costs.
12
+
3
13
  ## 0.8.0 - 2026-09-18
4
14
 
5
15
  - Add typed delegation to `SubAgentTool`: subclasses declare their own
data/README.md CHANGED
@@ -18,6 +18,10 @@ For GPT-6 Astra tools, opt into the pinned RubyLLM 2 release candidate and
18
18
  The [live validation app](examples/interactive_validation/README.md) exercises
19
19
  these controls with Rails 8.1, PostgreSQL, Sidekiq, and actual Astra/xhigh requests.
20
20
 
21
+ For non-generative judgments, use [structured evaluations](docs/structured-evaluations.md).
22
+ Cloudflare Jev supports Noul, Choice, and Score through a separate typed API with
23
+ durable receipts, usage/cost accounting, and output-audit integration—not chat messages.
24
+
21
25
  ## Installation
22
26
 
23
27
  Add this line to your application's **Gemfile**:
@@ -0,0 +1,76 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "net/http"
4
+ require "timeout"
5
+
6
+ module TurnKit
7
+ module Adapters
8
+ class CloudflareJev
9
+ MODEL = "typesafe/jev"
10
+
11
+ def initialize(account_id:, api_token:)
12
+ raise ConfigError, "Cloudflare account ID is required" unless account_id.to_s.match?(/\A[a-zA-Z0-9_-]+\z/)
13
+ raise ConfigError, "Cloudflare API token is required" if api_token.to_s.empty?
14
+ @account_id, @api_token = account_id, api_token
15
+ end
16
+
17
+ def inspect = "#<#{self.class.name}>"
18
+
19
+ # Include route/account, but never the bearer credential, in receipt identity.
20
+ def identity = { "provider" => "cloudflare", "account" => @account_id, "schema" => 1 }
21
+ def cost_model(model:) = "cloudflare/#{model}"
22
+
23
+ def validate!(model:, state:, questions:)
24
+ raise InputError, "unsupported Cloudflare evaluation model" unless model == MODEL
25
+ Evaluation.input!(state: state, questions: questions)
26
+ end
27
+
28
+ def evaluate(model:, state:, questions:, timeout:)
29
+ raise ArgumentError, "timeout must be positive and finite" unless timeout.is_a?(Numeric) && timeout.finite? && timeout.positive?
30
+ input = validate!(model: model, state: state, questions: questions)
31
+ uri = URI("https://api.cloudflare.com/client/v4/accounts/#{@account_id}/ai/run")
32
+ request = Net::HTTP::Post.new(uri)
33
+ request["Authorization"] = "Bearer #{@api_token}"
34
+ request["Content-Type"] = "application/json"
35
+ request.body = JSON.generate(model: model, input: input)
36
+ response = Timeout.timeout(timeout) do
37
+ Net::HTTP.start(uri.host, uri.port, use_ssl: true, open_timeout: timeout, read_timeout: timeout, write_timeout: timeout) do |http|
38
+ http.max_retries = 0
39
+ http.request(request)
40
+ end
41
+ end
42
+ code = response.code.to_i
43
+ unless code.between?(200, 299)
44
+ transient = [408, 429, 500, 502, 503, 504].include?(code)
45
+ # Quota exhaustion is not transient capacity pressure.
46
+ if code == 429
47
+ errors = JSON.parse(response.body).fetch("errors", []) rescue []
48
+ transient = false if Array(errors).any? { |error| error.is_a?(Hash) && error["code"] == 3036 }
49
+ end
50
+ delay = retry_delay(response["Retry-After"])
51
+ raise EvaluationError.new(code >= 500 || code == 408 ? :uncertain : :unavailable,
52
+ retryable: transient, retry_after: delay, http_status: code)
53
+ end
54
+ value = JSON.parse(response.body)
55
+ if value.is_a?(Hash) && value.key?("result")
56
+ raise EvaluationError.new(:unavailable) unless value["success"] == true
57
+ value = value["result"]
58
+ end
59
+ Evaluation.result!(value, questions: input.fetch("questions"))
60
+ rescue JSON::ParserError
61
+ raise EvaluationError.new(:malformed), cause: nil
62
+ rescue Timeout::Error, IOError, SystemCallError, OpenSSL::SSL::SSLError, SocketError, Net::HTTPBadResponse, Net::ProtocolError
63
+ raise EvaluationError.new(:uncertain, retryable: true), cause: nil
64
+ end
65
+
66
+ private
67
+ def retry_delay(value)
68
+ return unless value
69
+ return value.to_f if value.match?(/\A\d+(\.\d+)?\z/)
70
+ [Time.httpdate(value) - Clock.now, 0].max
71
+ rescue ArgumentError
72
+ nil
73
+ end
74
+ end
75
+ end
76
+ end
data/lib/turnkit/cost.rb CHANGED
@@ -10,6 +10,13 @@ module TurnKit
10
10
  def self.aggregate(costs)
11
11
  costs = costs.compact
12
12
  return new unless costs.any?
13
+ return new(unknown: true) if costs.any?(&:unknown?)
14
+ # A provider-supplied total cannot be assigned to a token component.
15
+ # Preserve it when combining chat totals with itemized evaluation prices.
16
+ if costs.any? { |cost| cost.total && COMPONENTS.all? { |component| cost.public_send(component).nil? } }
17
+ totals = costs.map(&:total)
18
+ return totals.any?(&:nil?) ? new : new(total: totals.sum)
19
+ end
13
20
 
14
21
  if costs.any? { |cost| COMPONENTS.any? { |component| !cost.public_send(component).nil? } }
15
22
  values = COMPONENTS.to_h do |component|
@@ -41,6 +48,10 @@ module TurnKit
41
48
 
42
49
  def self.from_record(record)
43
50
  attrs = record.transform_keys(&:to_s)
51
+ evaluations = attrs.dig("options", "state", "evaluations") || {}
52
+ if evaluations.values.any? { |receipt| receipt.fetch("attempts", []).any? { |attempt| attempt["cost"].nil? } }
53
+ return new(unknown: true)
54
+ end
44
55
  usage = attrs["usage"] || {}
45
56
  return from_hash(usage["cost_details"] || usage[:cost_details]) if usage["cost_details"] || usage[:cost_details]
46
57
  return new(total: attrs["cost"]) if attrs["cost"]
@@ -93,7 +104,8 @@ module TurnKit
93
104
  cache_read: hash[:cache_read],
94
105
  cache_write: hash[:cache_write],
95
106
  thinking: hash[:thinking],
96
- total: hash[:total]
107
+ total: hash[:total],
108
+ unknown: hash[:unknown] || false
97
109
  )
98
110
  end
99
111
 
@@ -120,7 +132,7 @@ module TurnKit
120
132
  tokens.to_i * price.to_f / PER_MILLION
121
133
  end
122
134
 
123
- def initialize(input: nil, output: nil, cache_read: nil, cache_write: nil, thinking: nil, total: nil, strict: false)
135
+ def initialize(input: nil, output: nil, cache_read: nil, cache_write: nil, thinking: nil, total: nil, strict: false, unknown: false)
124
136
  @input = number(input)
125
137
  @output = number(output)
126
138
  @cache_read = number(cache_read)
@@ -128,9 +140,13 @@ module TurnKit
128
140
  @thinking = number(thinking)
129
141
  @total = number(total)
130
142
  @strict = strict
143
+ @unknown = unknown
131
144
  end
132
145
 
146
+ def unknown? = @unknown
147
+
133
148
  def total
149
+ return nil if unknown?
134
150
  return @total if @total
135
151
  return nil if @strict && COMPONENTS.any? { |component| public_send(component).nil? }
136
152
 
@@ -145,7 +161,8 @@ module TurnKit
145
161
  "cache_read" => cache_read,
146
162
  "cache_write" => cache_write,
147
163
  "thinking" => thinking,
148
- "total" => total
164
+ "total" => total,
165
+ "unknown" => (true if unknown?)
149
166
  }.compact
150
167
  end
151
168
 
@@ -0,0 +1,119 @@
1
+ # frozen_string_literal: true
2
+
3
+ module TurnKit
4
+ # Deliberately separate from Result: evaluations have no message parts or text.
5
+ class EvaluationResult
6
+ attr_reader :model, :answers, :usage, :receipt_id
7
+
8
+ def initialize(model:, answers:, usage:, receipt_id: nil)
9
+ @model, @answers, @usage, @receipt_id = model, answers, usage, receipt_id
10
+ end
11
+
12
+ def to_h
13
+ { "model" => model, "answers" => answers, "usage" => usage.to_h }
14
+ end
15
+
16
+ def self.from_h(value, receipt_id: nil)
17
+ new(model: value.fetch("model"), answers: value.fetch("answers"),
18
+ usage: Usage.from_h(value.fetch("usage")), receipt_id: receipt_id)
19
+ end
20
+ end
21
+
22
+ class EvaluationError < Error
23
+ attr_reader :status, :retry_after, :usage, :model, :http_status
24
+
25
+ def initialize(status, retryable: false, retry_after: nil, usage: nil, model: nil, http_status: nil)
26
+ @status, @retryable, @retry_after, @usage, @model = status.to_s, retryable, retry_after, usage, model
27
+ @http_status = http_status
28
+ # Never include provider bodies, URLs, state, question names, or credentials.
29
+ super("evaluation #{@status}")
30
+ end
31
+
32
+ def retryable? = @retryable
33
+ end
34
+
35
+ module Evaluation
36
+ module_function
37
+
38
+ # Canonical JSON also rejects Ruby objects which JSON.generate would stringify.
39
+ def json(value)
40
+ case value
41
+ when Hash
42
+ raise InputError, "evaluation keys must be strings or symbols" unless value.keys.all? { |k| k.is_a?(String) || k.is_a?(Symbol) }
43
+ raise InputError, "duplicate evaluation keys" unless value.keys.map(&:to_s).uniq.size == value.size
44
+ value.transform_keys(&:to_s).sort.to_h.transform_values { |v| json(v) }
45
+ when Array then value.map { |v| json(v) }
46
+ when String, Integer, TrueClass, FalseClass, NilClass then value
47
+ when Float
48
+ raise InputError, "evaluation numbers must be finite" unless value.finite?
49
+ value
50
+ else raise InputError, "evaluation values must be JSON"
51
+ end
52
+ end
53
+
54
+ def text?(value) = value.nil? || value.is_a?(String) || value.is_a?(Hash) || value.is_a?(Array)
55
+
56
+ def input!(state:, questions:)
57
+ input = json({ state: state, questions: questions })
58
+ valid = text?(input["state"]) && input["questions"].is_a?(Hash)
59
+ raise InputError, "invalid evaluation input" unless valid
60
+ input["questions"].each do |id, question|
61
+ valid = !id.empty? && question.is_a?(Hash) && (question.keys - %w[type instructions criteria]).empty? &&
62
+ question.key?("instructions") && text?(question["instructions"])
63
+ raise InputError, "invalid evaluation question" unless valid
64
+ criteria = question["criteria"]
65
+ valid = case question["type"]
66
+ when "noul"
67
+ criteria.nil? || (criteria.is_a?(Hash) && (criteria.keys - %w[true false]).empty? && criteria.values.all? { |v| text?(v) })
68
+ when "choice"
69
+ criteria.is_a?(Hash) && criteria.values.all? { |v| text?(v) }
70
+ when "score"
71
+ criteria.is_a?(Array) && criteria.size >= 2 && criteria.all? { |v| text?(v) }
72
+ else false
73
+ end
74
+ raise InputError, "invalid evaluation criteria" unless valid
75
+ end
76
+ input
77
+ end
78
+
79
+ def probability?(value) = value.is_a?(Numeric) && value.finite? && value.between?(0, 1)
80
+
81
+ def usage(value)
82
+ return unless value.is_a?(Hash) && value.keys.sort == %w[input_tokens output_tokens]
83
+ return unless value.values.all? { |v| v.is_a?(Integer) && v.between?(0, 9_007_199_254_740_991) }
84
+ Usage.new(input_tokens: value["input_tokens"], output_tokens: value["output_tokens"])
85
+ end
86
+
87
+ def result!(value, questions:)
88
+ observed = usage(value["usage"]) if value.is_a?(Hash)
89
+ model = value["model"] if value.is_a?(Hash) && value["model"].is_a?(String) && !value["model"].empty?
90
+ invalid = -> { raise EvaluationError.new(:malformed, usage: observed, model: model) }
91
+ invalid.call unless value.is_a?(Hash) && value.keys.sort == %w[answers model usage] && model && observed
92
+ answers = value["answers"]
93
+ invalid.call unless answers.is_a?(Hash) && answers.keys.sort == questions.keys.sort
94
+ answers.each do |id, answer|
95
+ question = questions.fetch(id)
96
+ type = question.fetch("type")
97
+ invalid.call unless answer.is_a?(Hash) && answer["type"] == type
98
+ if type == "noul"
99
+ invalid.call unless answer.keys.sort == %w[noul type] && probability?(answer["noul"])
100
+ next
101
+ end
102
+ keys = type == "choice" ? %w[choice confidence probabilities type] : %w[confidence legend probabilities score type]
103
+ invalid.call unless answer.keys.sort == keys && probability?(answer["confidence"])
104
+ expected = type == "choice" ? question.fetch("criteria").keys : question.fetch("criteria").each_index.map(&:to_s)
105
+ probabilities = answer["probabilities"]
106
+ invalid.call unless probabilities.is_a?(Hash) && probabilities.keys.sort == expected.sort &&
107
+ probabilities.values.all? { |p| probability?(p) } && (probabilities.values.sum - 1).abs <= 0.01
108
+ if type == "choice"
109
+ invalid.call unless expected.include?(answer["choice"])
110
+ else
111
+ score, legend = answer.values_at("score", "legend")
112
+ invalid.call unless score.is_a?(Numeric) && score.finite? && score.between?(0, expected.size - 1) &&
113
+ legend.is_a?(Hash) && legend.keys.sort == expected.sort && legend.values.all? { |v| v.is_a?(String) }
114
+ end
115
+ end
116
+ EvaluationResult.new(model: model, answers: answers, usage: observed)
117
+ end
118
+ end
119
+ end
@@ -0,0 +1,142 @@
1
+ # frozen_string_literal: true
2
+
3
+ module TurnKit
4
+ module InternalEvaluation
5
+ # Call from a tool or output-audit callable, never from a persistence callback.
6
+ # before_dispatch is application transport/spend policy, not model input.
7
+ def internal_evaluation(evaluator:, model:, purpose:, state:, questions:, policy_version:,
8
+ candidate: nil, max_attempts: 1, timeout: 30, before_dispatch: nil)
9
+ raise LostClaim, "evaluation requires an owned execution" unless store.is_a?(ExecutionStore)
10
+ raise ArgumentError, "max_attempts must be 1..3" unless max_attempts.is_a?(Integer) && (1..3).include?(max_attempts)
11
+ raise ArgumentError, "timeout must be positive and finite" unless timeout.is_a?(Numeric) && timeout.finite? && timeout.positive?
12
+ input = evaluator.validate!(model: model, state: state, questions: questions)
13
+ reload
14
+ identity = Evaluation.json(provider: evaluator.identity, model: model, purpose: purpose.to_s,
15
+ policy_version: policy_version.to_s, candidate: candidate, input: input, schema: 1,
16
+ output_candidate: @record.dig("options", "state", "candidate"),
17
+ output_data: @record.dig("options", "state", "output_data"),
18
+ timeout: timeout, max_attempts: max_attempts)
19
+ key = Digest::SHA256.hexdigest(JSON.generate(identity))
20
+ cost_model = evaluator.cost_model(model: model)
21
+ deadline = Clock.now + timeout
22
+ loop do
23
+ receipt = store.atomic do
24
+ reload
25
+ raise LostClaim, "evaluation requires an owned execution" unless store.is_a?(ExecutionStore) && running?
26
+ (evaluation_receipts[key] || { "id" => key, "model" => model,
27
+ "provider" => identity["provider"], "purpose" => purpose.to_s,
28
+ "policy_version" => policy_version.to_s, "cost_model" => cost_model, "attempts" => [] })
29
+ end
30
+ return EvaluationResult.from_h(receipt.fetch("result"), receipt_id: key) if receipt["status"] == "completed"
31
+ last = receipt["attempts"].last
32
+ if last && (!last["retryable"] || receipt["attempts"].size >= max_attempts)
33
+ raise EvaluationError.new(last.fetch("status"), http_status: last["http_status"])
34
+ end
35
+ if last && last["owner"] == @record["claim_token"] && !last["finished_at"]
36
+ raise EvaluationError.new(:in_progress)
37
+ end
38
+ delay = last ? [last.fetch("retry_at", 0) - Clock.now.to_f, 0].max : 0
39
+ remaining = check_evaluation_dispatch!(deadline)
40
+ raise BudgetError, "evaluation deadline exceeded" if delay >= remaining
41
+ sleep(delay) if delay.positive?
42
+ Authorization.authorize!(:evaluate, principal: @record.dig("options", "principal"),
43
+ turn: self, evaluator: evaluator, model: model, purpose: purpose)
44
+ before_dispatch.call(turn: self, model: model, purpose: purpose) if before_dispatch
45
+ attempt = nil
46
+ remaining = nil
47
+ store.atomic do
48
+ reload
49
+ remaining = check_evaluation_dispatch!(deadline)
50
+ current = evaluation_receipts[key]
51
+ # A second caller may have committed or dispatched while policy ran.
52
+ raise EvaluationError.new(:in_progress) if current && current != receipt
53
+ attempt = { "id" => SecureRandom.uuid, "owner" => @record["claim_token"],
54
+ "status" => "uncertain", "retryable" => true, "started_at" => Clock.now.iso8601(6),
55
+ "cost" => nil }
56
+ receipt["attempts"] << attempt
57
+ receipt["status"] = "uncertain"
58
+ save_evaluation_receipt!(key, receipt)
59
+ end
60
+ event = { receipt_id: key, attempt_id: attempt["id"], model: model, purpose: purpose.to_s,
61
+ question_count: input.fetch("questions").size }
62
+ emit("evaluation.requested", event)
63
+ result = nil
64
+ failure = nil
65
+ begin
66
+ result = evaluator.evaluate(model: model, state: input.fetch("state"),
67
+ questions: input.fetch("questions"), timeout: remaining)
68
+ rescue EvaluationError => error
69
+ failure = error
70
+ end
71
+ observed_usage = result ? result.usage : failure.usage
72
+ observed_cost = Cost.from_usage(observed_usage, model: cost_model) if observed_usage
73
+ store.atomic do
74
+ reload
75
+ attempt.merge!("status" => result ? "completed" : failure.status,
76
+ "finished_at" => Clock.now.iso8601(6), "retryable" => failure&.retryable? || false,
77
+ "http_status" => failure&.http_status,
78
+ "usage" => observed_usage&.to_h, "cost" => observed_cost&.total,
79
+ "returned_model" => result ? result.model : failure.model)
80
+ if failure&.retryable?
81
+ attempt["retry_at"] = Clock.now.to_f + (failure.retry_after || rand * (2 ** (receipt["attempts"].size - 1)))
82
+ end
83
+ receipt["status"] = attempt["status"]
84
+ receipt["result"] = result.to_h if result
85
+ add_usage!(observed_usage, cost: observed_cost) if observed_usage
86
+ save_evaluation_receipt!(key, receipt)
87
+ end
88
+ emit(result ? "evaluation.completed" : "evaluation.failed", event.merge(
89
+ status: attempt["status"], returned_model: attempt["returned_model"], http_status: attempt["http_status"],
90
+ usage: observed_usage&.to_h, cost: observed_cost&.to_h,
91
+ duration: Clock.now - Time.iso8601(attempt["started_at"])))
92
+ # Keep the shared inline budget in sync without turning a paid successful
93
+ # assessment into a retry just because its own cost exhausted the budget.
94
+ begin
95
+ budget.add_cost!(observed_cost&.total)
96
+ rescue BudgetError
97
+ # The next dispatch checks the persisted root ledger.
98
+ end
99
+ root_budget = execution_budget
100
+ root_budget.check!(depth: depth, allow_exhausted_spend: true)
101
+ raise BudgetError, "evaluation deadline exceeded" if Clock.now >= deadline
102
+ return EvaluationResult.from_h(receipt.fetch("result"), receipt_id: key) if result
103
+ raise failure unless failure.retryable? && receipt["attempts"].size < max_attempts
104
+ end
105
+ end
106
+
107
+ # Copies, not references to live state. Attempts without usage have unknown
108
+ # cost, even if the turn's known-cost subtotal is zero.
109
+ def evaluation_receipts
110
+ JSON.parse(JSON.generate(@record.dig("options", "state", "evaluations") || {}))
111
+ end
112
+
113
+ def output_metadata = JSON.parse(JSON.generate(@record.dig("options", "state", "output_metadata") || {}))
114
+
115
+ def output_metadata=(value)
116
+ value = Evaluation.json(value)
117
+ raise InputError, "output metadata must be an object" unless value.is_a?(Hash)
118
+ raise LostClaim, "output metadata requires an owned execution" unless store.is_a?(ExecutionStore)
119
+ store.atomic do
120
+ reload
121
+ raise LostClaim, "output metadata requires an owned execution" unless store.is_a?(ExecutionStore) && running?
122
+ update_state!("output_metadata" => value)
123
+ end
124
+ end
125
+
126
+ private
127
+ def save_evaluation_receipt!(key, receipt)
128
+ update_state!("evaluations" => evaluation_receipts.merge(key => receipt))
129
+ end
130
+
131
+ def check_evaluation_dispatch!(deadline)
132
+ current_budget = execution_budget
133
+ current_budget.check!(depth: depth)
134
+ raise LostClaim, "evaluation execution is no longer running" unless running?
135
+ raise EvaluationError.new(:interrupted) if @record.dig("options", "controls", "pause_requested")
136
+ root_deadline = current_budget.root_started_at + current_budget.timeout if current_budget.timeout
137
+ remaining = [(deadline - Clock.now), (root_deadline - Clock.now if root_deadline)].compact.min
138
+ raise BudgetError, "evaluation deadline exceeded" unless remaining.positive?
139
+ remaining
140
+ end
141
+ end
142
+ end
@@ -32,6 +32,7 @@ module TurnKit
32
32
  def result(record)
33
33
  { "conversation_id" => record.fetch("conversation_id"), "turn_id" => record.fetch("id"),
34
34
  "status" => record.fetch("status"), "result" => record["output_text"].to_s,
35
+ "output_metadata" => record.dig("options", "state", "output_metadata"),
35
36
  "output_data" => record["output_data"], "error" => record["error"] }.compact
36
37
  end
37
38
  end
data/lib/turnkit/turn.rb CHANGED
@@ -3,6 +3,7 @@
3
3
  module TurnKit
4
4
  class Turn
5
5
  include TurnControls
6
+ include InternalEvaluation
6
7
  STATUSES = Record::TURN_STATUSES
7
8
 
8
9
  attr_reader :agent, :conversation, :store, :budget, :depth
@@ -325,6 +326,7 @@ module TurnKit
325
326
  add_usage!(result.usage, cost: cost)
326
327
  persist_assistant_message(result)
327
328
  update_state!("phase" => result.tool_calls? ? "tools" : "output", "parts" => result.parts,
329
+ "output_metadata" => nil,
328
330
  "candidate" => result.text, "output_data" => result.output_data, "terminal_tool_name" => nil,
329
331
  "budget_completion_call_id" => select_budget_completion(result))
330
332
  end
@@ -1,5 +1,5 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module TurnKit
4
- VERSION = "0.8.0"
4
+ VERSION = "0.9.0"
5
5
  end
data/lib/turnkit.rb CHANGED
@@ -46,9 +46,12 @@ require_relative "turnkit/load_skill_tool"
46
46
  require_relative "turnkit/message_projection"
47
47
  require_relative "turnkit/tool_runner"
48
48
  require_relative "turnkit/turn_controls"
49
+ require_relative "turnkit/evaluation"
50
+ require_relative "turnkit/internal_evaluation"
49
51
  require_relative "turnkit/turn"
50
52
  require_relative "turnkit/usage"
51
53
  require_relative "turnkit/run"
54
+ require_relative "turnkit/adapters/cloudflare_jev"
52
55
  require_relative "turnkit/adapters/codex"
53
56
  require_relative "turnkit/adapters/ruby_llm"
54
57
  require_relative "turnkit/active_record_store"
metadata CHANGED
@@ -1,14 +1,14 @@
1
1
  --- !ruby/object:Gem::Specification
2
2
  name: turnkit
3
3
  version: !ruby/object:Gem::Version
4
- version: 0.8.0
4
+ version: 0.9.0
5
5
  platform: ruby
6
6
  authors:
7
7
  - Sam Couch
8
8
  autorequire:
9
9
  bindir: bin
10
10
  cert_chain: []
11
- date: 2026-09-18 00:00:00.000000000 Z
11
+ date: 2026-09-19 00:00:00.000000000 Z
12
12
  dependencies: []
13
13
  description: TurnKit is a Ruby/Rails agent runtime for durable AI conversations, application
14
14
  runs, orchestrator agents, tool calling, skills, sub-agents, context compaction,
@@ -37,6 +37,7 @@ files:
37
37
  - lib/generators/turnkit/upgrade_generator.rb
38
38
  - lib/turnkit.rb
39
39
  - lib/turnkit/active_record_store.rb
40
+ - lib/turnkit/adapters/cloudflare_jev.rb
40
41
  - lib/turnkit/adapters/codex.rb
41
42
  - lib/turnkit/adapters/ruby_llm.rb
42
43
  - lib/turnkit/agent.rb
@@ -50,11 +51,13 @@ files:
50
51
  - lib/turnkit/coordination_tools.rb
51
52
  - lib/turnkit/cost.rb
52
53
  - lib/turnkit/error.rb
54
+ - lib/turnkit/evaluation.rb
53
55
  - lib/turnkit/event.rb
54
56
  - lib/turnkit/execution_store.rb
55
57
  - lib/turnkit/id.rb
56
58
  - lib/turnkit/image_result.rb
57
59
  - lib/turnkit/image_tool.rb
60
+ - lib/turnkit/internal_evaluation.rb
58
61
  - lib/turnkit/job.rb
59
62
  - lib/turnkit/load_skill_tool.rb
60
63
  - lib/turnkit/media_analysis_result.rb