turnkit 0.8.0 → 0.9.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +10 -0
- data/README.md +4 -0
- data/lib/turnkit/adapters/cloudflare_jev.rb +76 -0
- data/lib/turnkit/cost.rb +20 -3
- data/lib/turnkit/evaluation.rb +119 -0
- data/lib/turnkit/internal_evaluation.rb +142 -0
- data/lib/turnkit/sub_agent_tool.rb +1 -0
- data/lib/turnkit/turn.rb +2 -0
- data/lib/turnkit/version.rb +1 -1
- data/lib/turnkit.rb +3 -0
- metadata +5 -2
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: 73e85ee51e846b0d62828d964ad53b00fd0f6ce1e11d56611b47b45462d49d9f
|
|
4
|
+
data.tar.gz: f4c4591902e4a9a9fc4c2a50c6e768ad6a4191ac242418434ff5d8b22e99143b
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: cd39ba14aa1e48ce133d8b4cffc76d63939494a90466cfd62a5b70cb612f76f2caf8a5ffafc371272f00644ff1842a9b96768fea0724788721c45c3d36163655
|
|
7
|
+
data.tar.gz: a77ae5606a43be2790b54cf2a6c8eb78225b5f8b41d0f7b166b8c67b6568a357e65d1ae8f690b446c55afde370aab33d1d3f21dad30a7aec493eb2eb6007224c
|
data/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,15 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 0.9.0 - 2026-09-19
|
|
4
|
+
|
|
5
|
+
- Add `Turn#internal_evaluation` and a dependency-free Cloudflare Jev adapter for
|
|
6
|
+
Noul/Choice/Score judgments, separate from chat. Persist request/attempt receipts
|
|
7
|
+
with fenced atomic usage/cost accounting and bounded opt-in retries.
|
|
8
|
+
- Add optional runtime-owned `output_metadata` to child results so output-audit
|
|
9
|
+
callables can report assessments without changing strict generated schemas.
|
|
10
|
+
- Preserve unknown evaluation costs in cost summaries and retain total-only
|
|
11
|
+
provider charges when aggregating them with itemized costs.
|
|
12
|
+
|
|
3
13
|
## 0.8.0 - 2026-09-18
|
|
4
14
|
|
|
5
15
|
- Add typed delegation to `SubAgentTool`: subclasses declare their own
|
data/README.md
CHANGED
|
@@ -18,6 +18,10 @@ For GPT-6 Astra tools, opt into the pinned RubyLLM 2 release candidate and
|
|
|
18
18
|
The [live validation app](examples/interactive_validation/README.md) exercises
|
|
19
19
|
these controls with Rails 8.1, PostgreSQL, Sidekiq, and actual Astra/xhigh requests.
|
|
20
20
|
|
|
21
|
+
For non-generative judgments, use [structured evaluations](docs/structured-evaluations.md).
|
|
22
|
+
Cloudflare Jev supports Noul, Choice, and Score through a separate typed API with
|
|
23
|
+
durable receipts, usage/cost accounting, and output-audit integration—not chat messages.
|
|
24
|
+
|
|
21
25
|
## Installation
|
|
22
26
|
|
|
23
27
|
Add this line to your application's **Gemfile**:
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "net/http"
|
|
4
|
+
require "timeout"
|
|
5
|
+
|
|
6
|
+
module TurnKit
|
|
7
|
+
module Adapters
|
|
8
|
+
class CloudflareJev
|
|
9
|
+
MODEL = "typesafe/jev"
|
|
10
|
+
|
|
11
|
+
def initialize(account_id:, api_token:)
|
|
12
|
+
raise ConfigError, "Cloudflare account ID is required" unless account_id.to_s.match?(/\A[a-zA-Z0-9_-]+\z/)
|
|
13
|
+
raise ConfigError, "Cloudflare API token is required" if api_token.to_s.empty?
|
|
14
|
+
@account_id, @api_token = account_id, api_token
|
|
15
|
+
end
|
|
16
|
+
|
|
17
|
+
def inspect = "#<#{self.class.name}>"
|
|
18
|
+
|
|
19
|
+
# Include route/account, but never the bearer credential, in receipt identity.
|
|
20
|
+
def identity = { "provider" => "cloudflare", "account" => @account_id, "schema" => 1 }
|
|
21
|
+
def cost_model(model:) = "cloudflare/#{model}"
|
|
22
|
+
|
|
23
|
+
def validate!(model:, state:, questions:)
|
|
24
|
+
raise InputError, "unsupported Cloudflare evaluation model" unless model == MODEL
|
|
25
|
+
Evaluation.input!(state: state, questions: questions)
|
|
26
|
+
end
|
|
27
|
+
|
|
28
|
+
def evaluate(model:, state:, questions:, timeout:)
|
|
29
|
+
raise ArgumentError, "timeout must be positive and finite" unless timeout.is_a?(Numeric) && timeout.finite? && timeout.positive?
|
|
30
|
+
input = validate!(model: model, state: state, questions: questions)
|
|
31
|
+
uri = URI("https://api.cloudflare.com/client/v4/accounts/#{@account_id}/ai/run")
|
|
32
|
+
request = Net::HTTP::Post.new(uri)
|
|
33
|
+
request["Authorization"] = "Bearer #{@api_token}"
|
|
34
|
+
request["Content-Type"] = "application/json"
|
|
35
|
+
request.body = JSON.generate(model: model, input: input)
|
|
36
|
+
response = Timeout.timeout(timeout) do
|
|
37
|
+
Net::HTTP.start(uri.host, uri.port, use_ssl: true, open_timeout: timeout, read_timeout: timeout, write_timeout: timeout) do |http|
|
|
38
|
+
http.max_retries = 0
|
|
39
|
+
http.request(request)
|
|
40
|
+
end
|
|
41
|
+
end
|
|
42
|
+
code = response.code.to_i
|
|
43
|
+
unless code.between?(200, 299)
|
|
44
|
+
transient = [408, 429, 500, 502, 503, 504].include?(code)
|
|
45
|
+
# Quota exhaustion is not transient capacity pressure.
|
|
46
|
+
if code == 429
|
|
47
|
+
errors = JSON.parse(response.body).fetch("errors", []) rescue []
|
|
48
|
+
transient = false if Array(errors).any? { |error| error.is_a?(Hash) && error["code"] == 3036 }
|
|
49
|
+
end
|
|
50
|
+
delay = retry_delay(response["Retry-After"])
|
|
51
|
+
raise EvaluationError.new(code >= 500 || code == 408 ? :uncertain : :unavailable,
|
|
52
|
+
retryable: transient, retry_after: delay, http_status: code)
|
|
53
|
+
end
|
|
54
|
+
value = JSON.parse(response.body)
|
|
55
|
+
if value.is_a?(Hash) && value.key?("result")
|
|
56
|
+
raise EvaluationError.new(:unavailable) unless value["success"] == true
|
|
57
|
+
value = value["result"]
|
|
58
|
+
end
|
|
59
|
+
Evaluation.result!(value, questions: input.fetch("questions"))
|
|
60
|
+
rescue JSON::ParserError
|
|
61
|
+
raise EvaluationError.new(:malformed), cause: nil
|
|
62
|
+
rescue Timeout::Error, IOError, SystemCallError, OpenSSL::SSL::SSLError, SocketError, Net::HTTPBadResponse, Net::ProtocolError
|
|
63
|
+
raise EvaluationError.new(:uncertain, retryable: true), cause: nil
|
|
64
|
+
end
|
|
65
|
+
|
|
66
|
+
private
|
|
67
|
+
def retry_delay(value)
|
|
68
|
+
return unless value
|
|
69
|
+
return value.to_f if value.match?(/\A\d+(\.\d+)?\z/)
|
|
70
|
+
[Time.httpdate(value) - Clock.now, 0].max
|
|
71
|
+
rescue ArgumentError
|
|
72
|
+
nil
|
|
73
|
+
end
|
|
74
|
+
end
|
|
75
|
+
end
|
|
76
|
+
end
|
data/lib/turnkit/cost.rb
CHANGED
|
@@ -10,6 +10,13 @@ module TurnKit
|
|
|
10
10
|
def self.aggregate(costs)
|
|
11
11
|
costs = costs.compact
|
|
12
12
|
return new unless costs.any?
|
|
13
|
+
return new(unknown: true) if costs.any?(&:unknown?)
|
|
14
|
+
# A provider-supplied total cannot be assigned to a token component.
|
|
15
|
+
# Preserve it when combining chat totals with itemized evaluation prices.
|
|
16
|
+
if costs.any? { |cost| cost.total && COMPONENTS.all? { |component| cost.public_send(component).nil? } }
|
|
17
|
+
totals = costs.map(&:total)
|
|
18
|
+
return totals.any?(&:nil?) ? new : new(total: totals.sum)
|
|
19
|
+
end
|
|
13
20
|
|
|
14
21
|
if costs.any? { |cost| COMPONENTS.any? { |component| !cost.public_send(component).nil? } }
|
|
15
22
|
values = COMPONENTS.to_h do |component|
|
|
@@ -41,6 +48,10 @@ module TurnKit
|
|
|
41
48
|
|
|
42
49
|
def self.from_record(record)
|
|
43
50
|
attrs = record.transform_keys(&:to_s)
|
|
51
|
+
evaluations = attrs.dig("options", "state", "evaluations") || {}
|
|
52
|
+
if evaluations.values.any? { |receipt| receipt.fetch("attempts", []).any? { |attempt| attempt["cost"].nil? } }
|
|
53
|
+
return new(unknown: true)
|
|
54
|
+
end
|
|
44
55
|
usage = attrs["usage"] || {}
|
|
45
56
|
return from_hash(usage["cost_details"] || usage[:cost_details]) if usage["cost_details"] || usage[:cost_details]
|
|
46
57
|
return new(total: attrs["cost"]) if attrs["cost"]
|
|
@@ -93,7 +104,8 @@ module TurnKit
|
|
|
93
104
|
cache_read: hash[:cache_read],
|
|
94
105
|
cache_write: hash[:cache_write],
|
|
95
106
|
thinking: hash[:thinking],
|
|
96
|
-
total: hash[:total]
|
|
107
|
+
total: hash[:total],
|
|
108
|
+
unknown: hash[:unknown] || false
|
|
97
109
|
)
|
|
98
110
|
end
|
|
99
111
|
|
|
@@ -120,7 +132,7 @@ module TurnKit
|
|
|
120
132
|
tokens.to_i * price.to_f / PER_MILLION
|
|
121
133
|
end
|
|
122
134
|
|
|
123
|
-
def initialize(input: nil, output: nil, cache_read: nil, cache_write: nil, thinking: nil, total: nil, strict: false)
|
|
135
|
+
def initialize(input: nil, output: nil, cache_read: nil, cache_write: nil, thinking: nil, total: nil, strict: false, unknown: false)
|
|
124
136
|
@input = number(input)
|
|
125
137
|
@output = number(output)
|
|
126
138
|
@cache_read = number(cache_read)
|
|
@@ -128,9 +140,13 @@ module TurnKit
|
|
|
128
140
|
@thinking = number(thinking)
|
|
129
141
|
@total = number(total)
|
|
130
142
|
@strict = strict
|
|
143
|
+
@unknown = unknown
|
|
131
144
|
end
|
|
132
145
|
|
|
146
|
+
def unknown? = @unknown
|
|
147
|
+
|
|
133
148
|
def total
|
|
149
|
+
return nil if unknown?
|
|
134
150
|
return @total if @total
|
|
135
151
|
return nil if @strict && COMPONENTS.any? { |component| public_send(component).nil? }
|
|
136
152
|
|
|
@@ -145,7 +161,8 @@ module TurnKit
|
|
|
145
161
|
"cache_read" => cache_read,
|
|
146
162
|
"cache_write" => cache_write,
|
|
147
163
|
"thinking" => thinking,
|
|
148
|
-
"total" => total
|
|
164
|
+
"total" => total,
|
|
165
|
+
"unknown" => (true if unknown?)
|
|
149
166
|
}.compact
|
|
150
167
|
end
|
|
151
168
|
|
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module TurnKit
|
|
4
|
+
# Deliberately separate from Result: evaluations have no message parts or text.
|
|
5
|
+
class EvaluationResult
|
|
6
|
+
attr_reader :model, :answers, :usage, :receipt_id
|
|
7
|
+
|
|
8
|
+
def initialize(model:, answers:, usage:, receipt_id: nil)
|
|
9
|
+
@model, @answers, @usage, @receipt_id = model, answers, usage, receipt_id
|
|
10
|
+
end
|
|
11
|
+
|
|
12
|
+
def to_h
|
|
13
|
+
{ "model" => model, "answers" => answers, "usage" => usage.to_h }
|
|
14
|
+
end
|
|
15
|
+
|
|
16
|
+
def self.from_h(value, receipt_id: nil)
|
|
17
|
+
new(model: value.fetch("model"), answers: value.fetch("answers"),
|
|
18
|
+
usage: Usage.from_h(value.fetch("usage")), receipt_id: receipt_id)
|
|
19
|
+
end
|
|
20
|
+
end
|
|
21
|
+
|
|
22
|
+
class EvaluationError < Error
|
|
23
|
+
attr_reader :status, :retry_after, :usage, :model, :http_status
|
|
24
|
+
|
|
25
|
+
def initialize(status, retryable: false, retry_after: nil, usage: nil, model: nil, http_status: nil)
|
|
26
|
+
@status, @retryable, @retry_after, @usage, @model = status.to_s, retryable, retry_after, usage, model
|
|
27
|
+
@http_status = http_status
|
|
28
|
+
# Never include provider bodies, URLs, state, question names, or credentials.
|
|
29
|
+
super("evaluation #{@status}")
|
|
30
|
+
end
|
|
31
|
+
|
|
32
|
+
def retryable? = @retryable
|
|
33
|
+
end
|
|
34
|
+
|
|
35
|
+
module Evaluation
|
|
36
|
+
module_function
|
|
37
|
+
|
|
38
|
+
# Canonical JSON also rejects Ruby objects which JSON.generate would stringify.
|
|
39
|
+
def json(value)
|
|
40
|
+
case value
|
|
41
|
+
when Hash
|
|
42
|
+
raise InputError, "evaluation keys must be strings or symbols" unless value.keys.all? { |k| k.is_a?(String) || k.is_a?(Symbol) }
|
|
43
|
+
raise InputError, "duplicate evaluation keys" unless value.keys.map(&:to_s).uniq.size == value.size
|
|
44
|
+
value.transform_keys(&:to_s).sort.to_h.transform_values { |v| json(v) }
|
|
45
|
+
when Array then value.map { |v| json(v) }
|
|
46
|
+
when String, Integer, TrueClass, FalseClass, NilClass then value
|
|
47
|
+
when Float
|
|
48
|
+
raise InputError, "evaluation numbers must be finite" unless value.finite?
|
|
49
|
+
value
|
|
50
|
+
else raise InputError, "evaluation values must be JSON"
|
|
51
|
+
end
|
|
52
|
+
end
|
|
53
|
+
|
|
54
|
+
def text?(value) = value.nil? || value.is_a?(String) || value.is_a?(Hash) || value.is_a?(Array)
|
|
55
|
+
|
|
56
|
+
def input!(state:, questions:)
|
|
57
|
+
input = json({ state: state, questions: questions })
|
|
58
|
+
valid = text?(input["state"]) && input["questions"].is_a?(Hash)
|
|
59
|
+
raise InputError, "invalid evaluation input" unless valid
|
|
60
|
+
input["questions"].each do |id, question|
|
|
61
|
+
valid = !id.empty? && question.is_a?(Hash) && (question.keys - %w[type instructions criteria]).empty? &&
|
|
62
|
+
question.key?("instructions") && text?(question["instructions"])
|
|
63
|
+
raise InputError, "invalid evaluation question" unless valid
|
|
64
|
+
criteria = question["criteria"]
|
|
65
|
+
valid = case question["type"]
|
|
66
|
+
when "noul"
|
|
67
|
+
criteria.nil? || (criteria.is_a?(Hash) && (criteria.keys - %w[true false]).empty? && criteria.values.all? { |v| text?(v) })
|
|
68
|
+
when "choice"
|
|
69
|
+
criteria.is_a?(Hash) && criteria.values.all? { |v| text?(v) }
|
|
70
|
+
when "score"
|
|
71
|
+
criteria.is_a?(Array) && criteria.size >= 2 && criteria.all? { |v| text?(v) }
|
|
72
|
+
else false
|
|
73
|
+
end
|
|
74
|
+
raise InputError, "invalid evaluation criteria" unless valid
|
|
75
|
+
end
|
|
76
|
+
input
|
|
77
|
+
end
|
|
78
|
+
|
|
79
|
+
def probability?(value) = value.is_a?(Numeric) && value.finite? && value.between?(0, 1)
|
|
80
|
+
|
|
81
|
+
def usage(value)
|
|
82
|
+
return unless value.is_a?(Hash) && value.keys.sort == %w[input_tokens output_tokens]
|
|
83
|
+
return unless value.values.all? { |v| v.is_a?(Integer) && v.between?(0, 9_007_199_254_740_991) }
|
|
84
|
+
Usage.new(input_tokens: value["input_tokens"], output_tokens: value["output_tokens"])
|
|
85
|
+
end
|
|
86
|
+
|
|
87
|
+
def result!(value, questions:)
|
|
88
|
+
observed = usage(value["usage"]) if value.is_a?(Hash)
|
|
89
|
+
model = value["model"] if value.is_a?(Hash) && value["model"].is_a?(String) && !value["model"].empty?
|
|
90
|
+
invalid = -> { raise EvaluationError.new(:malformed, usage: observed, model: model) }
|
|
91
|
+
invalid.call unless value.is_a?(Hash) && value.keys.sort == %w[answers model usage] && model && observed
|
|
92
|
+
answers = value["answers"]
|
|
93
|
+
invalid.call unless answers.is_a?(Hash) && answers.keys.sort == questions.keys.sort
|
|
94
|
+
answers.each do |id, answer|
|
|
95
|
+
question = questions.fetch(id)
|
|
96
|
+
type = question.fetch("type")
|
|
97
|
+
invalid.call unless answer.is_a?(Hash) && answer["type"] == type
|
|
98
|
+
if type == "noul"
|
|
99
|
+
invalid.call unless answer.keys.sort == %w[noul type] && probability?(answer["noul"])
|
|
100
|
+
next
|
|
101
|
+
end
|
|
102
|
+
keys = type == "choice" ? %w[choice confidence probabilities type] : %w[confidence legend probabilities score type]
|
|
103
|
+
invalid.call unless answer.keys.sort == keys && probability?(answer["confidence"])
|
|
104
|
+
expected = type == "choice" ? question.fetch("criteria").keys : question.fetch("criteria").each_index.map(&:to_s)
|
|
105
|
+
probabilities = answer["probabilities"]
|
|
106
|
+
invalid.call unless probabilities.is_a?(Hash) && probabilities.keys.sort == expected.sort &&
|
|
107
|
+
probabilities.values.all? { |p| probability?(p) } && (probabilities.values.sum - 1).abs <= 0.01
|
|
108
|
+
if type == "choice"
|
|
109
|
+
invalid.call unless expected.include?(answer["choice"])
|
|
110
|
+
else
|
|
111
|
+
score, legend = answer.values_at("score", "legend")
|
|
112
|
+
invalid.call unless score.is_a?(Numeric) && score.finite? && score.between?(0, expected.size - 1) &&
|
|
113
|
+
legend.is_a?(Hash) && legend.keys.sort == expected.sort && legend.values.all? { |v| v.is_a?(String) }
|
|
114
|
+
end
|
|
115
|
+
end
|
|
116
|
+
EvaluationResult.new(model: model, answers: answers, usage: observed)
|
|
117
|
+
end
|
|
118
|
+
end
|
|
119
|
+
end
|
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module TurnKit
|
|
4
|
+
module InternalEvaluation
|
|
5
|
+
# Call from a tool or output-audit callable, never from a persistence callback.
|
|
6
|
+
# before_dispatch is application transport/spend policy, not model input.
|
|
7
|
+
def internal_evaluation(evaluator:, model:, purpose:, state:, questions:, policy_version:,
|
|
8
|
+
candidate: nil, max_attempts: 1, timeout: 30, before_dispatch: nil)
|
|
9
|
+
raise LostClaim, "evaluation requires an owned execution" unless store.is_a?(ExecutionStore)
|
|
10
|
+
raise ArgumentError, "max_attempts must be 1..3" unless max_attempts.is_a?(Integer) && (1..3).include?(max_attempts)
|
|
11
|
+
raise ArgumentError, "timeout must be positive and finite" unless timeout.is_a?(Numeric) && timeout.finite? && timeout.positive?
|
|
12
|
+
input = evaluator.validate!(model: model, state: state, questions: questions)
|
|
13
|
+
reload
|
|
14
|
+
identity = Evaluation.json(provider: evaluator.identity, model: model, purpose: purpose.to_s,
|
|
15
|
+
policy_version: policy_version.to_s, candidate: candidate, input: input, schema: 1,
|
|
16
|
+
output_candidate: @record.dig("options", "state", "candidate"),
|
|
17
|
+
output_data: @record.dig("options", "state", "output_data"),
|
|
18
|
+
timeout: timeout, max_attempts: max_attempts)
|
|
19
|
+
key = Digest::SHA256.hexdigest(JSON.generate(identity))
|
|
20
|
+
cost_model = evaluator.cost_model(model: model)
|
|
21
|
+
deadline = Clock.now + timeout
|
|
22
|
+
loop do
|
|
23
|
+
receipt = store.atomic do
|
|
24
|
+
reload
|
|
25
|
+
raise LostClaim, "evaluation requires an owned execution" unless store.is_a?(ExecutionStore) && running?
|
|
26
|
+
(evaluation_receipts[key] || { "id" => key, "model" => model,
|
|
27
|
+
"provider" => identity["provider"], "purpose" => purpose.to_s,
|
|
28
|
+
"policy_version" => policy_version.to_s, "cost_model" => cost_model, "attempts" => [] })
|
|
29
|
+
end
|
|
30
|
+
return EvaluationResult.from_h(receipt.fetch("result"), receipt_id: key) if receipt["status"] == "completed"
|
|
31
|
+
last = receipt["attempts"].last
|
|
32
|
+
if last && (!last["retryable"] || receipt["attempts"].size >= max_attempts)
|
|
33
|
+
raise EvaluationError.new(last.fetch("status"), http_status: last["http_status"])
|
|
34
|
+
end
|
|
35
|
+
if last && last["owner"] == @record["claim_token"] && !last["finished_at"]
|
|
36
|
+
raise EvaluationError.new(:in_progress)
|
|
37
|
+
end
|
|
38
|
+
delay = last ? [last.fetch("retry_at", 0) - Clock.now.to_f, 0].max : 0
|
|
39
|
+
remaining = check_evaluation_dispatch!(deadline)
|
|
40
|
+
raise BudgetError, "evaluation deadline exceeded" if delay >= remaining
|
|
41
|
+
sleep(delay) if delay.positive?
|
|
42
|
+
Authorization.authorize!(:evaluate, principal: @record.dig("options", "principal"),
|
|
43
|
+
turn: self, evaluator: evaluator, model: model, purpose: purpose)
|
|
44
|
+
before_dispatch.call(turn: self, model: model, purpose: purpose) if before_dispatch
|
|
45
|
+
attempt = nil
|
|
46
|
+
remaining = nil
|
|
47
|
+
store.atomic do
|
|
48
|
+
reload
|
|
49
|
+
remaining = check_evaluation_dispatch!(deadline)
|
|
50
|
+
current = evaluation_receipts[key]
|
|
51
|
+
# A second caller may have committed or dispatched while policy ran.
|
|
52
|
+
raise EvaluationError.new(:in_progress) if current && current != receipt
|
|
53
|
+
attempt = { "id" => SecureRandom.uuid, "owner" => @record["claim_token"],
|
|
54
|
+
"status" => "uncertain", "retryable" => true, "started_at" => Clock.now.iso8601(6),
|
|
55
|
+
"cost" => nil }
|
|
56
|
+
receipt["attempts"] << attempt
|
|
57
|
+
receipt["status"] = "uncertain"
|
|
58
|
+
save_evaluation_receipt!(key, receipt)
|
|
59
|
+
end
|
|
60
|
+
event = { receipt_id: key, attempt_id: attempt["id"], model: model, purpose: purpose.to_s,
|
|
61
|
+
question_count: input.fetch("questions").size }
|
|
62
|
+
emit("evaluation.requested", event)
|
|
63
|
+
result = nil
|
|
64
|
+
failure = nil
|
|
65
|
+
begin
|
|
66
|
+
result = evaluator.evaluate(model: model, state: input.fetch("state"),
|
|
67
|
+
questions: input.fetch("questions"), timeout: remaining)
|
|
68
|
+
rescue EvaluationError => error
|
|
69
|
+
failure = error
|
|
70
|
+
end
|
|
71
|
+
observed_usage = result ? result.usage : failure.usage
|
|
72
|
+
observed_cost = Cost.from_usage(observed_usage, model: cost_model) if observed_usage
|
|
73
|
+
store.atomic do
|
|
74
|
+
reload
|
|
75
|
+
attempt.merge!("status" => result ? "completed" : failure.status,
|
|
76
|
+
"finished_at" => Clock.now.iso8601(6), "retryable" => failure&.retryable? || false,
|
|
77
|
+
"http_status" => failure&.http_status,
|
|
78
|
+
"usage" => observed_usage&.to_h, "cost" => observed_cost&.total,
|
|
79
|
+
"returned_model" => result ? result.model : failure.model)
|
|
80
|
+
if failure&.retryable?
|
|
81
|
+
attempt["retry_at"] = Clock.now.to_f + (failure.retry_after || rand * (2 ** (receipt["attempts"].size - 1)))
|
|
82
|
+
end
|
|
83
|
+
receipt["status"] = attempt["status"]
|
|
84
|
+
receipt["result"] = result.to_h if result
|
|
85
|
+
add_usage!(observed_usage, cost: observed_cost) if observed_usage
|
|
86
|
+
save_evaluation_receipt!(key, receipt)
|
|
87
|
+
end
|
|
88
|
+
emit(result ? "evaluation.completed" : "evaluation.failed", event.merge(
|
|
89
|
+
status: attempt["status"], returned_model: attempt["returned_model"], http_status: attempt["http_status"],
|
|
90
|
+
usage: observed_usage&.to_h, cost: observed_cost&.to_h,
|
|
91
|
+
duration: Clock.now - Time.iso8601(attempt["started_at"])))
|
|
92
|
+
# Keep the shared inline budget in sync without turning a paid successful
|
|
93
|
+
# assessment into a retry just because its own cost exhausted the budget.
|
|
94
|
+
begin
|
|
95
|
+
budget.add_cost!(observed_cost&.total)
|
|
96
|
+
rescue BudgetError
|
|
97
|
+
# The next dispatch checks the persisted root ledger.
|
|
98
|
+
end
|
|
99
|
+
root_budget = execution_budget
|
|
100
|
+
root_budget.check!(depth: depth, allow_exhausted_spend: true)
|
|
101
|
+
raise BudgetError, "evaluation deadline exceeded" if Clock.now >= deadline
|
|
102
|
+
return EvaluationResult.from_h(receipt.fetch("result"), receipt_id: key) if result
|
|
103
|
+
raise failure unless failure.retryable? && receipt["attempts"].size < max_attempts
|
|
104
|
+
end
|
|
105
|
+
end
|
|
106
|
+
|
|
107
|
+
# Copies, not references to live state. Attempts without usage have unknown
|
|
108
|
+
# cost, even if the turn's known-cost subtotal is zero.
|
|
109
|
+
def evaluation_receipts
|
|
110
|
+
JSON.parse(JSON.generate(@record.dig("options", "state", "evaluations") || {}))
|
|
111
|
+
end
|
|
112
|
+
|
|
113
|
+
def output_metadata = JSON.parse(JSON.generate(@record.dig("options", "state", "output_metadata") || {}))
|
|
114
|
+
|
|
115
|
+
def output_metadata=(value)
|
|
116
|
+
value = Evaluation.json(value)
|
|
117
|
+
raise InputError, "output metadata must be an object" unless value.is_a?(Hash)
|
|
118
|
+
raise LostClaim, "output metadata requires an owned execution" unless store.is_a?(ExecutionStore)
|
|
119
|
+
store.atomic do
|
|
120
|
+
reload
|
|
121
|
+
raise LostClaim, "output metadata requires an owned execution" unless store.is_a?(ExecutionStore) && running?
|
|
122
|
+
update_state!("output_metadata" => value)
|
|
123
|
+
end
|
|
124
|
+
end
|
|
125
|
+
|
|
126
|
+
private
|
|
127
|
+
def save_evaluation_receipt!(key, receipt)
|
|
128
|
+
update_state!("evaluations" => evaluation_receipts.merge(key => receipt))
|
|
129
|
+
end
|
|
130
|
+
|
|
131
|
+
def check_evaluation_dispatch!(deadline)
|
|
132
|
+
current_budget = execution_budget
|
|
133
|
+
current_budget.check!(depth: depth)
|
|
134
|
+
raise LostClaim, "evaluation execution is no longer running" unless running?
|
|
135
|
+
raise EvaluationError.new(:interrupted) if @record.dig("options", "controls", "pause_requested")
|
|
136
|
+
root_deadline = current_budget.root_started_at + current_budget.timeout if current_budget.timeout
|
|
137
|
+
remaining = [(deadline - Clock.now), (root_deadline - Clock.now if root_deadline)].compact.min
|
|
138
|
+
raise BudgetError, "evaluation deadline exceeded" unless remaining.positive?
|
|
139
|
+
remaining
|
|
140
|
+
end
|
|
141
|
+
end
|
|
142
|
+
end
|
|
@@ -32,6 +32,7 @@ module TurnKit
|
|
|
32
32
|
def result(record)
|
|
33
33
|
{ "conversation_id" => record.fetch("conversation_id"), "turn_id" => record.fetch("id"),
|
|
34
34
|
"status" => record.fetch("status"), "result" => record["output_text"].to_s,
|
|
35
|
+
"output_metadata" => record.dig("options", "state", "output_metadata"),
|
|
35
36
|
"output_data" => record["output_data"], "error" => record["error"] }.compact
|
|
36
37
|
end
|
|
37
38
|
end
|
data/lib/turnkit/turn.rb
CHANGED
|
@@ -3,6 +3,7 @@
|
|
|
3
3
|
module TurnKit
|
|
4
4
|
class Turn
|
|
5
5
|
include TurnControls
|
|
6
|
+
include InternalEvaluation
|
|
6
7
|
STATUSES = Record::TURN_STATUSES
|
|
7
8
|
|
|
8
9
|
attr_reader :agent, :conversation, :store, :budget, :depth
|
|
@@ -325,6 +326,7 @@ module TurnKit
|
|
|
325
326
|
add_usage!(result.usage, cost: cost)
|
|
326
327
|
persist_assistant_message(result)
|
|
327
328
|
update_state!("phase" => result.tool_calls? ? "tools" : "output", "parts" => result.parts,
|
|
329
|
+
"output_metadata" => nil,
|
|
328
330
|
"candidate" => result.text, "output_data" => result.output_data, "terminal_tool_name" => nil,
|
|
329
331
|
"budget_completion_call_id" => select_budget_completion(result))
|
|
330
332
|
end
|
data/lib/turnkit/version.rb
CHANGED
data/lib/turnkit.rb
CHANGED
|
@@ -46,9 +46,12 @@ require_relative "turnkit/load_skill_tool"
|
|
|
46
46
|
require_relative "turnkit/message_projection"
|
|
47
47
|
require_relative "turnkit/tool_runner"
|
|
48
48
|
require_relative "turnkit/turn_controls"
|
|
49
|
+
require_relative "turnkit/evaluation"
|
|
50
|
+
require_relative "turnkit/internal_evaluation"
|
|
49
51
|
require_relative "turnkit/turn"
|
|
50
52
|
require_relative "turnkit/usage"
|
|
51
53
|
require_relative "turnkit/run"
|
|
54
|
+
require_relative "turnkit/adapters/cloudflare_jev"
|
|
52
55
|
require_relative "turnkit/adapters/codex"
|
|
53
56
|
require_relative "turnkit/adapters/ruby_llm"
|
|
54
57
|
require_relative "turnkit/active_record_store"
|
metadata
CHANGED
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
--- !ruby/object:Gem::Specification
|
|
2
2
|
name: turnkit
|
|
3
3
|
version: !ruby/object:Gem::Version
|
|
4
|
-
version: 0.
|
|
4
|
+
version: 0.9.0
|
|
5
5
|
platform: ruby
|
|
6
6
|
authors:
|
|
7
7
|
- Sam Couch
|
|
8
8
|
autorequire:
|
|
9
9
|
bindir: bin
|
|
10
10
|
cert_chain: []
|
|
11
|
-
date: 2026-09-
|
|
11
|
+
date: 2026-09-19 00:00:00.000000000 Z
|
|
12
12
|
dependencies: []
|
|
13
13
|
description: TurnKit is a Ruby/Rails agent runtime for durable AI conversations, application
|
|
14
14
|
runs, orchestrator agents, tool calling, skills, sub-agents, context compaction,
|
|
@@ -37,6 +37,7 @@ files:
|
|
|
37
37
|
- lib/generators/turnkit/upgrade_generator.rb
|
|
38
38
|
- lib/turnkit.rb
|
|
39
39
|
- lib/turnkit/active_record_store.rb
|
|
40
|
+
- lib/turnkit/adapters/cloudflare_jev.rb
|
|
40
41
|
- lib/turnkit/adapters/codex.rb
|
|
41
42
|
- lib/turnkit/adapters/ruby_llm.rb
|
|
42
43
|
- lib/turnkit/agent.rb
|
|
@@ -50,11 +51,13 @@ files:
|
|
|
50
51
|
- lib/turnkit/coordination_tools.rb
|
|
51
52
|
- lib/turnkit/cost.rb
|
|
52
53
|
- lib/turnkit/error.rb
|
|
54
|
+
- lib/turnkit/evaluation.rb
|
|
53
55
|
- lib/turnkit/event.rb
|
|
54
56
|
- lib/turnkit/execution_store.rb
|
|
55
57
|
- lib/turnkit/id.rb
|
|
56
58
|
- lib/turnkit/image_result.rb
|
|
57
59
|
- lib/turnkit/image_tool.rb
|
|
60
|
+
- lib/turnkit/internal_evaluation.rb
|
|
58
61
|
- lib/turnkit/job.rb
|
|
59
62
|
- lib/turnkit/load_skill_tool.rb
|
|
60
63
|
- lib/turnkit/media_analysis_result.rb
|