@pi-in-go/pigpen-jev 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CREDITS.md +22 -0
- package/LICENSE +22 -0
- package/README.md +237 -0
- package/extensions/jev/ask.go +166 -0
- package/extensions/jev/ask_test.go +218 -0
- package/extensions/jev/backend.go +128 -0
- package/extensions/jev/bench_test.go +64 -0
- package/extensions/jev/boundaries_test.go +159 -0
- package/extensions/jev/command.go +224 -0
- package/extensions/jev/commands_test.go +214 -0
- package/extensions/jev/config.go +450 -0
- package/extensions/jev/errors_test.go +191 -0
- package/extensions/jev/extension.go +391 -0
- package/extensions/jev/fakehost_test.go +548 -0
- package/extensions/jev/gate.go +125 -0
- package/extensions/jev/gate_test.go +610 -0
- package/extensions/jev/gatekey_test.go +24 -0
- package/extensions/jev/go.mod +9 -0
- package/extensions/jev/go.sum +2 -0
- package/extensions/jev/go.work +10 -0
- package/extensions/jev/helpers_test.go +404 -0
- package/extensions/jev/memo.go +88 -0
- package/extensions/jev/output.go +89 -0
- package/extensions/jev/output_test.go +187 -0
- package/extensions/jev/ownmodel_test.go +118 -0
- package/extensions/jev/render.go +136 -0
- package/extensions/jev/review_test.go +310 -0
- package/extensions/jev/source_test.go +57 -0
- package/extensions/jev/text.go +174 -0
- package/extensions/jev/trust_test.go +335 -0
- package/extensions/jev/types.go +227 -0
- package/libs/typesafe/CONTRACT.md +125 -0
- package/libs/typesafe/CREDITS.md +37 -0
- package/libs/typesafe/LICENSE +23 -0
- package/libs/typesafe/README.md +19 -0
- package/libs/typesafe/go.mod +3 -0
- package/libs/typesafe/libraries/ownmodel/backend_test.go +496 -0
- package/libs/typesafe/libraries/ownmodel/canon.go +190 -0
- package/libs/typesafe/libraries/ownmodel/convert.go +199 -0
- package/libs/typesafe/libraries/ownmodel/doc.go +15 -0
- package/libs/typesafe/libraries/ownmodel/equivalence_test.go +199 -0
- package/libs/typesafe/libraries/ownmodel/helpers_test.go +155 -0
- package/libs/typesafe/libraries/ownmodel/mutation_test.go +31 -0
- package/libs/typesafe/libraries/ownmodel/ownmodel.go +225 -0
- package/libs/typesafe/libraries/ownmodel/plan.go +442 -0
- package/libs/typesafe/libraries/ownmodel/run.go +288 -0
- package/libs/typesafe/libraries/ownmodel/schema_test.go +254 -0
- package/libs/typesafe/libraries/ownmodel/twins_test.go +169 -0
- package/libs/typesafe/libraries/ownmodel/utils_test.go +125 -0
- package/libs/typesafe/libraries/pigmodel/pigmodel.go +264 -0
- package/libs/typesafe/libraries/pigmodel/pigmodel_test.go +410 -0
- package/libs/typesafe/libraries/typesafe/answers.go +268 -0
- package/libs/typesafe/libraries/typesafe/api_response_test.go +113 -0
- package/libs/typesafe/libraries/typesafe/batch.go +80 -0
- package/libs/typesafe/libraries/typesafe/batch_test.go +133 -0
- package/libs/typesafe/libraries/typesafe/bench_test.go +71 -0
- package/libs/typesafe/libraries/typesafe/client.go +561 -0
- package/libs/typesafe/libraries/typesafe/client_test.go +495 -0
- package/libs/typesafe/libraries/typesafe/crosscheck_test.go +464 -0
- package/libs/typesafe/libraries/typesafe/crosscheck_workflowevals_test.go +219 -0
- package/libs/typesafe/libraries/typesafe/doc.go +27 -0
- package/libs/typesafe/libraries/typesafe/entry.go +142 -0
- package/libs/typesafe/libraries/typesafe/env.go +11 -0
- package/libs/typesafe/libraries/typesafe/errors.go +310 -0
- package/libs/typesafe/libraries/typesafe/errors_test.go +175 -0
- package/libs/typesafe/libraries/typesafe/helpers_test.go +294 -0
- package/libs/typesafe/libraries/typesafe/live_test.go +96 -0
- package/libs/typesafe/libraries/typesafe/logging.go +160 -0
- package/libs/typesafe/libraries/typesafe/logging_test.go +259 -0
- package/libs/typesafe/libraries/typesafe/marshal_test.go +112 -0
- package/libs/typesafe/libraries/typesafe/mutation_test.go +39 -0
- package/libs/typesafe/libraries/typesafe/questions.go +490 -0
- package/libs/typesafe/libraries/typesafe/questions_test.go +166 -0
- package/libs/typesafe/libraries/typesafe/regressions_test.go +159 -0
- package/libs/typesafe/libraries/typesafe/reliability_test.go +649 -0
- package/libs/typesafe/libraries/typesafe/retry.go +350 -0
- package/libs/typesafe/libraries/typesafe/retry_test.go +297 -0
- package/libs/typesafe/libraries/typesafe/runtime_test.go +26 -0
- package/libs/typesafe/libraries/typesafe/transport_test.go +163 -0
- package/libs/typesafe/libraries/typesafe/twins_test.go +127 -0
- package/libs/typesafe/libraries/typesafe/types_test.go +165 -0
- package/libs/typesafe/libraries/typesafe/version.go +10 -0
- package/libs/typesafe/package.json +37 -0
- package/libs/typesafe/provenance.json +49 -0
- package/package.json +42 -0
- package/port/PORT.md +107 -0
- package/port/e2e/gate-and-output.py +35 -0
- package/port/e2e/jev-ask.py +36 -0
- package/port/e2e/model-switch.py +44 -0
- package/port/e2e/off-by-default.py +34 -0
- package/port/gen-scenarios.py +103 -0
- package/port/golden/cache-identical-calls.jsonl +30 -0
- package/port/golden/clear.jsonl +22 -0
- package/port/golden/commands.jsonl +43 -0
- package/port/golden/enforce-accept.jsonl +23 -0
- package/port/golden/enforce-decline.jsonl +22 -0
- package/port/golden/jev-ask.jsonl +20 -0
- package/port/golden/output-advice.jsonl +23 -0
- package/port/golden/output-leak.jsonl +24 -0
- package/port/golden/output-low-confidence.jsonl +22 -0
- package/port/golden/shadow-flagged.jsonl +23 -0
- package/port/golden/unjudged-tools.jsonl +19 -0
- package/port/golden/write-elision.jsonl +21 -0
- package/port/mutate-unit.py +63 -0
- package/port/mutations.json +578 -0
- package/port/oracle/LICENSE +21 -0
- package/port/oracle/README.md +181 -0
- package/port/oracle/SHA256SUMS +8 -0
- package/port/oracle/package.json +43 -0
- package/port/oracle/src/client.ts +409 -0
- package/port/oracle/src/config.ts +363 -0
- package/port/oracle/src/gate.ts +229 -0
- package/port/oracle/src/index.ts +649 -0
- package/port/oracle/src/output.ts +163 -0
- package/port/red-run.log +309 -0
- package/port/scenarios/cache-identical-calls.json +71 -0
- package/port/scenarios/clear.json +61 -0
- package/port/scenarios/commands.json +119 -0
- package/port/scenarios/enforce-accept.json +66 -0
- package/port/scenarios/enforce-decline.json +57 -0
- package/port/scenarios/jev-ask.json +83 -0
- package/port/scenarios/output-advice.json +61 -0
- package/port/scenarios/output-leak.json +61 -0
- package/port/scenarios/output-low-confidence.json +61 -0
- package/port/scenarios/shadow-flagged.json +61 -0
- package/port/scenarios/unjudged-tools.json +55 -0
- package/port/scenarios/write-elision.json +53 -0
- package/provenance.json +18 -0
|
@@ -0,0 +1,155 @@
|
|
|
1
|
+
package ownmodel
|
|
2
|
+
|
|
3
|
+
import (
|
|
4
|
+
"context"
|
|
5
|
+
"encoding/json"
|
|
6
|
+
"errors"
|
|
7
|
+
"net/http"
|
|
8
|
+
"reflect"
|
|
9
|
+
"strings"
|
|
10
|
+
"sync"
|
|
11
|
+
"testing"
|
|
12
|
+
"time"
|
|
13
|
+
|
|
14
|
+
"github.com/MichaelKinsy/pigpen/components/typesafe/libraries/typesafe"
|
|
15
|
+
)
|
|
16
|
+
|
|
17
|
+
// twin marks a test as the twin of upstream cases: ids are pytest node ids from
|
|
18
|
+
// port/twins/system-one-adapter-python.txt. skipTwin records cases with no Go counterpart.
|
|
19
|
+
func twin(t testing.TB, ids ...string) { t.Helper() }
|
|
20
|
+
|
|
21
|
+
func skipTwin(t testing.TB, reason string, ids ...string) {
|
|
22
|
+
t.Helper()
|
|
23
|
+
t.Skip(reason)
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
func ptr[T any](v T) *T { return &v }
|
|
27
|
+
|
|
28
|
+
// scripted is the analog of the oracle's _ScriptedProvider: it answers with a scripted
|
|
29
|
+
// sequence of payloads (encoded to JSON), raw strings, Results and errors; the last step
|
|
30
|
+
// repeats once the script is exhausted.
|
|
31
|
+
type scripted struct {
|
|
32
|
+
mu sync.Mutex
|
|
33
|
+
steps []any
|
|
34
|
+
usage [2]int
|
|
35
|
+
calls [][]Message
|
|
36
|
+
structured []bool
|
|
37
|
+
schemas []map[string]any
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
func newScripted(steps ...any) *scripted { return &scripted{steps: steps, usage: [2]int{11, 7}} }
|
|
41
|
+
|
|
42
|
+
func (s *scripted) Name() string { return "fake-model" }
|
|
43
|
+
|
|
44
|
+
func (s *scripted) Complete(ctx context.Context, req Request) (Result, error) {
|
|
45
|
+
s.mu.Lock()
|
|
46
|
+
defer s.mu.Unlock()
|
|
47
|
+
s.calls = append(s.calls, append([]Message(nil), req.Messages...))
|
|
48
|
+
s.structured = append(s.structured, req.Structured)
|
|
49
|
+
s.schemas = append(s.schemas, req.Schema)
|
|
50
|
+
i := len(s.calls) - 1
|
|
51
|
+
if i >= len(s.steps) {
|
|
52
|
+
i = len(s.steps) - 1
|
|
53
|
+
}
|
|
54
|
+
switch step := s.steps[i].(type) {
|
|
55
|
+
case error:
|
|
56
|
+
return Result{}, step
|
|
57
|
+
case Result:
|
|
58
|
+
return step, nil
|
|
59
|
+
case string:
|
|
60
|
+
return Result{Text: step, InputTokens: ptr(s.usage[0]), OutputTokens: ptr(s.usage[1])}, nil
|
|
61
|
+
default:
|
|
62
|
+
raw, err := json.Marshal(step)
|
|
63
|
+
if err != nil {
|
|
64
|
+
panic(err)
|
|
65
|
+
}
|
|
66
|
+
return Result{Text: string(raw), InputTokens: ptr(s.usage[0]), OutputTokens: ptr(s.usage[1])}, nil
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
func (s *scripted) callCount() int {
|
|
71
|
+
s.mu.Lock()
|
|
72
|
+
defer s.mu.Unlock()
|
|
73
|
+
return len(s.calls)
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
func (s *scripted) lastCall() []Message {
|
|
77
|
+
s.mu.Lock()
|
|
78
|
+
defer s.mu.Unlock()
|
|
79
|
+
return s.calls[len(s.calls)-1]
|
|
80
|
+
}
|
|
81
|
+
|
|
82
|
+
// providerError is the error a real model source returns after translating an HTTP failure.
|
|
83
|
+
func providerError(status int) error {
|
|
84
|
+
return typesafe.NewAPIError(status, map[string]any{"message": "unavailable"}, http.Header{})
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
func mustNew(t testing.TB, opts Options) *Backend {
|
|
88
|
+
t.Helper()
|
|
89
|
+
b, err := New(opts)
|
|
90
|
+
if err != nil {
|
|
91
|
+
t.Fatalf("New: %v", err)
|
|
92
|
+
}
|
|
93
|
+
return b
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
func evaluate(t testing.TB, b *Backend, state any, qs typesafe.Questions, opts *typesafe.RequestOptions) (*Evaluation, error) {
|
|
97
|
+
t.Helper()
|
|
98
|
+
return b.Evaluate(context.Background(), typesafe.SystemOneRequest{State: typesafe.EntryOf(state), Questions: qs}, opts)
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
func eq[T any](t testing.TB, got, want T) {
|
|
102
|
+
t.Helper()
|
|
103
|
+
if !reflect.DeepEqual(got, want) {
|
|
104
|
+
t.Fatalf("got %#v, want %#v", got, want)
|
|
105
|
+
}
|
|
106
|
+
}
|
|
107
|
+
|
|
108
|
+
func noErr(t testing.TB, err error) {
|
|
109
|
+
t.Helper()
|
|
110
|
+
if err != nil {
|
|
111
|
+
t.Fatalf("unexpected error %T: %v", err, err)
|
|
112
|
+
}
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
func contains(t testing.TB, s, sub string) {
|
|
116
|
+
t.Helper()
|
|
117
|
+
if !strings.Contains(s, sub) {
|
|
118
|
+
t.Fatalf("%q does not contain %q", s, sub)
|
|
119
|
+
}
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
func mustAs[T any](t testing.TB, err error) T {
|
|
123
|
+
t.Helper()
|
|
124
|
+
var target T
|
|
125
|
+
if !errors.As(err, &target) {
|
|
126
|
+
t.Fatalf("error %T (%v) is not %T", err, err, target)
|
|
127
|
+
}
|
|
128
|
+
return target
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
func questionSet() typesafe.Questions {
|
|
132
|
+
return typesafe.Questions{
|
|
133
|
+
typesafe.Ask("positive", typesafe.Noul("The review is positive.")),
|
|
134
|
+
typesafe.Ask("stars", typesafe.Score("Rating.", "Bad.", "Good.")),
|
|
135
|
+
typesafe.Ask("genre", typesafe.Choice("Genre.", typesafe.Opt("fiction", "A story."), typesafe.Opt("nonfiction", "Facts."))),
|
|
136
|
+
}
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
func answerNoul(name string) typesafe.Questions {
|
|
140
|
+
return typesafe.Questions{typesafe.Ask(name, typesafe.Noul("The review is positive."))}
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
func categories(reasons []RetryReason) []string {
|
|
144
|
+
out := []string{}
|
|
145
|
+
for _, r := range reasons {
|
|
146
|
+
out = append(out, r.Category)
|
|
147
|
+
}
|
|
148
|
+
return out
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
func fastRetry(max int) typesafe.RetryOverrides {
|
|
152
|
+
return typesafe.RetryOverrides{MaxRetries: typesafe.Ptr(max), BackoffInitial: typesafe.Ptr(durMS(1)), BackoffJitter: typesafe.Ptr(0.0)}
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
func durMS(n int) time.Duration { return time.Duration(n) * time.Millisecond }
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
package ownmodel
|
|
2
|
+
|
|
3
|
+
import "testing"
|
|
4
|
+
|
|
5
|
+
// Added by the mutation check (port/mutations.json: tolerance-loosened): the oracle's tolerance
|
|
6
|
+
// is 1e-6, so a deviation just above it is normalized and just below it is left alone.
|
|
7
|
+
func TestNormalizationToleranceIsOneMillionth(t *testing.T) {
|
|
8
|
+
answers := []string{"a", "b"}
|
|
9
|
+
over := normalizeProbabilities(answers, map[string]float64{"a": 0.500005, "b": 0.500005}, "", Probabilities, true)
|
|
10
|
+
approx(t, over.Probabilities["a"], 0.5)
|
|
11
|
+
if over.Original == nil {
|
|
12
|
+
t.Fatal("a deviation of 1e-5 must be normalized and its original kept")
|
|
13
|
+
}
|
|
14
|
+
under := normalizeProbabilities(answers, map[string]float64{"a": 0.5000004, "b": 0.5000004}, "", Probabilities, true)
|
|
15
|
+
approx(t, under.Probabilities["a"], 0.5000004)
|
|
16
|
+
if under.Original != nil {
|
|
17
|
+
t.Fatal("a deviation of 8e-7 is within the tolerance")
|
|
18
|
+
}
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
// Added by the mutation check (default-transient-retries-2): the oracle's default is no
|
|
22
|
+
// transient retries, so a failing provider is called once unless the caller opts in.
|
|
23
|
+
func TestNoTransientRetriesByDefault(t *testing.T) {
|
|
24
|
+
model := newScripted(providerError(500))
|
|
25
|
+
b := mustNew(t, Options{Model: model})
|
|
26
|
+
_, err := evaluate(t, b, "state", answerNoul("positive"), nil)
|
|
27
|
+
if err == nil {
|
|
28
|
+
t.Fatal("expected the provider error")
|
|
29
|
+
}
|
|
30
|
+
eq(t, model.callCount(), 1)
|
|
31
|
+
}
|
|
@@ -0,0 +1,225 @@
|
|
|
1
|
+
package ownmodel
|
|
2
|
+
|
|
3
|
+
import (
|
|
4
|
+
"context"
|
|
5
|
+
"time"
|
|
6
|
+
|
|
7
|
+
"github.com/MichaelKinsy/pigpen/components/typesafe/libraries/typesafe"
|
|
8
|
+
)
|
|
9
|
+
|
|
10
|
+
// Role is the author of a message.
|
|
11
|
+
type Role string
|
|
12
|
+
|
|
13
|
+
// Message roles.
|
|
14
|
+
const (
|
|
15
|
+
RoleSystem Role = "system"
|
|
16
|
+
RoleUser Role = "user"
|
|
17
|
+
RoleAssistant Role = "assistant"
|
|
18
|
+
)
|
|
19
|
+
|
|
20
|
+
// Message is one chat message in provider-neutral form.
|
|
21
|
+
type Message struct {
|
|
22
|
+
Role Role `json:"role"`
|
|
23
|
+
Content string `json:"content"`
|
|
24
|
+
}
|
|
25
|
+
|
|
26
|
+
// Request is one model request: the conversation and the JSON Schema of the answer.
|
|
27
|
+
// When Structured is false the schema is also in the system prompt and the model is
|
|
28
|
+
// only asked for JSON text (prompted mode).
|
|
29
|
+
type Request struct {
|
|
30
|
+
Messages []Message
|
|
31
|
+
Schema map[string]any
|
|
32
|
+
Structured bool
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
// Result is the raw text the model produced and its token counts; nil counts are
|
|
36
|
+
// unreported.
|
|
37
|
+
type Result struct {
|
|
38
|
+
Text string `json:"text"`
|
|
39
|
+
InputTokens *int `json:"input_tokens"`
|
|
40
|
+
OutputTokens *int `json:"output_tokens"`
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
// Model performs one model request. It is the seam between this package and a model
|
|
44
|
+
// source; implementations return typesafe errors so the retry policy can classify
|
|
45
|
+
// them: *typesafe.APIError (with a status) and *typesafe.APIConnectionError are
|
|
46
|
+
// retried per policy, anything else is final. A refusal, an incomplete generation or
|
|
47
|
+
// an output-limit stop must be a plain *typesafe.TypeSafeError (never retried and
|
|
48
|
+
// never treated as malformed output).
|
|
49
|
+
type Model interface {
|
|
50
|
+
// Name is the model name reported in the result.
|
|
51
|
+
Name() string
|
|
52
|
+
Complete(ctx context.Context, req Request) (Result, error)
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
// AnswerMode selects what the model is asked to return.
|
|
56
|
+
type AnswerMode string
|
|
57
|
+
|
|
58
|
+
// Answer modes.
|
|
59
|
+
const (
|
|
60
|
+
// Probabilities asks for a probability per label (a probability for a Noul).
|
|
61
|
+
Probabilities AnswerMode = "probabilities"
|
|
62
|
+
// Discrete asks for exactly one value per question.
|
|
63
|
+
Discrete AnswerMode = "discrete"
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
// Options configure a [Backend].
|
|
67
|
+
type Options struct {
|
|
68
|
+
// Model answers the questions. Required unless Resolve is set.
|
|
69
|
+
Model Model
|
|
70
|
+
// Resolve, when set, picks the Model for a request that names one
|
|
71
|
+
// (SystemOneRequest.Model); a request that names none uses Model.
|
|
72
|
+
Resolve func(name string) (Model, error)
|
|
73
|
+
// AnswerMode is Probabilities (the default when empty) or Discrete.
|
|
74
|
+
AnswerMode AnswerMode
|
|
75
|
+
// StructuredOutputs asks the Model for its native structured-output mode; when false
|
|
76
|
+
// the schema goes in the prompt and the JSON is validated here.
|
|
77
|
+
StructuredOutputs bool
|
|
78
|
+
// NormalizeProbabilities rescales invalid probability distributions to sum to 1.
|
|
79
|
+
NormalizeProbabilities bool
|
|
80
|
+
// MalformedRetries is the number of corrective retries when the output fails
|
|
81
|
+
// validation. Default 0.
|
|
82
|
+
MalformedRetries int
|
|
83
|
+
// Retry overrides the transient-error policy, applied over a policy with no retries
|
|
84
|
+
// (the default is no retries, as in the Python adapter).
|
|
85
|
+
Retry typesafe.RetryOverrides
|
|
86
|
+
// Logger receives request summaries; nil discards.
|
|
87
|
+
Logger typesafe.Logger
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
// Backend answers questions with a Model. It is safe for concurrent use.
|
|
91
|
+
type Backend struct {
|
|
92
|
+
opts Options
|
|
93
|
+
mode AnswerMode
|
|
94
|
+
policy typesafe.RetryPolicy
|
|
95
|
+
logger typesafe.Logger
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
var _ typesafe.Evaluator = (*Backend)(nil)
|
|
99
|
+
|
|
100
|
+
// New validates the options. It returns a *typesafe.TypeSafeError for an unknown
|
|
101
|
+
// answer mode, a negative MalformedRetries, or no Model and no Resolve.
|
|
102
|
+
func New(opts Options) (*Backend, error) {
|
|
103
|
+
mode := opts.AnswerMode
|
|
104
|
+
if mode == "" {
|
|
105
|
+
mode = Probabilities
|
|
106
|
+
}
|
|
107
|
+
if mode != Probabilities && mode != Discrete {
|
|
108
|
+
return nil, &typesafe.TypeSafeError{Message: "AnswerMode must be 'probabilities' or 'discrete'."}
|
|
109
|
+
}
|
|
110
|
+
if opts.MalformedRetries < 0 {
|
|
111
|
+
return nil, &typesafe.TypeSafeError{Message: "MalformedRetries must be >= 0."}
|
|
112
|
+
}
|
|
113
|
+
if opts.Model == nil && opts.Resolve == nil {
|
|
114
|
+
return nil, &typesafe.TypeSafeError{Message: "An LLM model is required: set Options.Model (or Options.Resolve)."}
|
|
115
|
+
}
|
|
116
|
+
base := typesafe.DefaultRetryPolicy()
|
|
117
|
+
base.MaxRetries = 0 // the oracle's default: no transient retries
|
|
118
|
+
policy, err := base.Resolve(opts.Retry)
|
|
119
|
+
if err != nil {
|
|
120
|
+
return nil, err
|
|
121
|
+
}
|
|
122
|
+
logger := opts.Logger
|
|
123
|
+
if logger == nil {
|
|
124
|
+
logger = discardLogger{}
|
|
125
|
+
}
|
|
126
|
+
return &Backend{opts: opts, mode: mode, policy: policy, logger: logger}, nil
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
type discardLogger struct{}
|
|
130
|
+
|
|
131
|
+
func (discardLogger) Debug(string, ...any) {}
|
|
132
|
+
func (discardLogger) Info(string, ...any) {}
|
|
133
|
+
func (discardLogger) Warn(string, ...any) {}
|
|
134
|
+
func (discardLogger) Error(string, ...any) {}
|
|
135
|
+
|
|
136
|
+
// SystemOne implements [typesafe.Evaluator]: the questions are validated, one prompt is
|
|
137
|
+
// built from the state, the model is called (with corrective and transient retries),
|
|
138
|
+
// and its output is converted to typed answers. Usage token counts the model did not
|
|
139
|
+
// report are zero here; Evaluate keeps them unknown.
|
|
140
|
+
func (b *Backend) SystemOne(ctx context.Context, req typesafe.SystemOneRequest, opts *typesafe.RequestOptions) (*typesafe.SystemOneResult, error) {
|
|
141
|
+
ev, err := b.Evaluate(ctx, req, opts)
|
|
142
|
+
if err != nil {
|
|
143
|
+
return nil, err
|
|
144
|
+
}
|
|
145
|
+
return ev.Result, nil
|
|
146
|
+
}
|
|
147
|
+
|
|
148
|
+
// Usage is the token and retry accounting of one evaluation. A nil count is unknown;
|
|
149
|
+
// a total is nil if any attempt did not report that count.
|
|
150
|
+
type Usage struct {
|
|
151
|
+
InputTokens *int
|
|
152
|
+
OutputTokens *int
|
|
153
|
+
InputTokensTotal *int
|
|
154
|
+
OutputTokensTotal *int
|
|
155
|
+
// Retries counts transient-error retries, MalformedRetries corrective retries.
|
|
156
|
+
Retries int
|
|
157
|
+
MalformedRetries int
|
|
158
|
+
Latency time.Duration
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
// RetryReason records why one retry happened.
|
|
162
|
+
type RetryReason struct {
|
|
163
|
+
// Category is "provider_error" or "malformed_structure".
|
|
164
|
+
Category string
|
|
165
|
+
Message string
|
|
166
|
+
}
|
|
167
|
+
|
|
168
|
+
// Attempt records one model call, including failed ones.
|
|
169
|
+
type Attempt struct {
|
|
170
|
+
Messages []Message
|
|
171
|
+
Schema map[string]any
|
|
172
|
+
Structured bool
|
|
173
|
+
// Response is nil when the call failed before returning.
|
|
174
|
+
Response *Result
|
|
175
|
+
ModelName string
|
|
176
|
+
// Error and ErrorType describe a failed call.
|
|
177
|
+
Error string
|
|
178
|
+
ErrorType string
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
// Debug holds the diagnostics of one evaluation.
|
|
182
|
+
type Debug struct {
|
|
183
|
+
LLMAttempts []Attempt
|
|
184
|
+
RetryReasons []RetryReason
|
|
185
|
+
// MaxError is the largest deviation of a probability sum from 1; InvalidProbs
|
|
186
|
+
// counts the questions beyond the 1e-6 tolerance, ProbabilityErrors names them.
|
|
187
|
+
MaxError float64
|
|
188
|
+
InvalidProbs int
|
|
189
|
+
ProbabilityErrors map[string]float64
|
|
190
|
+
// OriginalProbabilities holds distributions that normalization changed.
|
|
191
|
+
OriginalProbabilities map[string]map[string]float64
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
// Evaluation is a result with usage and diagnostics.
|
|
195
|
+
type Evaluation struct {
|
|
196
|
+
Result *typesafe.SystemOneResult
|
|
197
|
+
Usage Usage
|
|
198
|
+
Debug Debug
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
// Evaluate is [Backend.SystemOne] with usage and diagnostics. An input error (no
|
|
202
|
+
// questions, a bad question, a missing model) is a plain *typesafe.TypeSafeError; a
|
|
203
|
+
// failure after the model was called is a *DebugError carrying the attempt history.
|
|
204
|
+
func (b *Backend) Evaluate(ctx context.Context, req typesafe.SystemOneRequest, opts *typesafe.RequestOptions) (*Evaluation, error) {
|
|
205
|
+
return b.evaluate(ctx, req, opts)
|
|
206
|
+
}
|
|
207
|
+
|
|
208
|
+
// MalformedOutputError is returned (inside a *DebugError) when the model's output still
|
|
209
|
+
// fails validation after the last corrective retry. Message is
|
|
210
|
+
// "Model output did not match the schema: <details>" and Cause the validation error.
|
|
211
|
+
type MalformedOutputError struct{ *typesafe.TypeSafeError }
|
|
212
|
+
|
|
213
|
+
func (e *MalformedOutputError) Unwrap() error { return e.TypeSafeError }
|
|
214
|
+
|
|
215
|
+
// DebugError carries the attempt history of a failed evaluation and unwraps to the
|
|
216
|
+
// cause: a *typesafe.TypeSafeError for malformed output that is still invalid after the
|
|
217
|
+
// last corrective retry (message "Model output did not match the schema: ..."), or the
|
|
218
|
+
// error the Model returned.
|
|
219
|
+
type DebugError struct {
|
|
220
|
+
Err error
|
|
221
|
+
Debug Debug
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
func (e *DebugError) Error() string { return e.Err.Error() }
|
|
225
|
+
func (e *DebugError) Unwrap() error { return e.Err }
|