@alvin0/ai-agent-sdk-decision-adapter 0.0.0-stage → 0.1.10
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +288 -2
- package/USE_CASES.md +217 -0
- package/dist/async-BPqlUFnU.js +53 -0
- package/dist/async-BPqlUFnU.js.map +1 -0
- package/dist/index.d.ts +224 -0
- package/dist/index.d.ts.map +1 -0
- package/dist/index.js +948 -0
- package/dist/index.js.map +1 -0
- package/dist/transport.d.ts +7 -0
- package/dist/transport.d.ts.map +1 -0
- package/dist/transport.js +3 -0
- package/package.json +72 -3
package/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 alvin0 (chaulamdinhai) <chaulamdinhai@gmail.com>
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
package/README.md
CHANGED
|
@@ -1,3 +1,289 @@
|
|
|
1
|
-
#
|
|
1
|
+
# Decision adapter
|
|
2
2
|
|
|
3
|
-
|
|
3
|
+
Runtime: **Universal** (Node, browser, and Web-standard workers).
|
|
4
|
+
|
|
5
|
+
```sh
|
|
6
|
+
pnpm add @alvin0/ai-agent-sdk-core @alvin0/ai-agent-sdk-decision-adapter
|
|
7
|
+
```
|
|
8
|
+
|
|
9
|
+
Provider-neutral typed decisions, independently of chat generation and embeddings.
|
|
10
|
+
Use `choiceQuestion`, `scoreQuestion`, and `booleanQuestion` to define a closed
|
|
11
|
+
answer space. The return type preserves question IDs and literal choice options.
|
|
12
|
+
|
|
13
|
+
See [Use cases and setup](./USE_CASES.md) for routing, triage, guardrails, RAG
|
|
14
|
+
reranking, extraction, and staged workflows, including their tradeoffs.
|
|
15
|
+
|
|
16
|
+
```ts
|
|
17
|
+
import {
|
|
18
|
+
createDecisionRuntime, choiceQuestion,
|
|
19
|
+
} from '@alvin0/ai-agent-sdk-decision-adapter'
|
|
20
|
+
import { typesafePlugin } from '@alvin0/ai-agent-sdk-provider-typesafe'
|
|
21
|
+
|
|
22
|
+
const runtime = createDecisionRuntime({
|
|
23
|
+
providers: [typesafePlugin({ apiKey: 'your-server-side-key' })],
|
|
24
|
+
})
|
|
25
|
+
try {
|
|
26
|
+
const model = runtime.decisionModel({ provider: 'typesafe', model: 'jev-latest' })
|
|
27
|
+
const result = await model.evaluate({
|
|
28
|
+
state: { message: 'Please refund my invoice' },
|
|
29
|
+
questions: {
|
|
30
|
+
department: choiceQuestion('Which department should handle this?', {
|
|
31
|
+
billing: 'Invoices and refunds', support: 'Technical issues',
|
|
32
|
+
}),
|
|
33
|
+
},
|
|
34
|
+
})
|
|
35
|
+
// Typed as 'billing' | 'support'.
|
|
36
|
+
console.log(result.answers.department.choice)
|
|
37
|
+
} finally {
|
|
38
|
+
await runtime.close()
|
|
39
|
+
}
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
## Contract
|
|
43
|
+
|
|
44
|
+
- `choice`: selected option; optional full probability distribution.
|
|
45
|
+
- `score`: expected zero-based ordinal level; fractional scores are valid.
|
|
46
|
+
- `boolean`: boolean value; optional probability of true. The TypeSafe provider
|
|
47
|
+
uses `probabilityTrue >= 0.5` for `value`; application thresholds can use the raw
|
|
48
|
+
probability instead.
|
|
49
|
+
- Evidence carries `probabilitySource`: `provider`, `token-logprobs`, or
|
|
50
|
+
`model-generated`. The tag identifies provenance, not a calibration guarantee.
|
|
51
|
+
Missing evidence/usage stays absent. Confidence is not inferred.
|
|
52
|
+
- JSON is detached and frozen before async preparation. Results are checked for
|
|
53
|
+
IDs, types, allowed choices, probability bounds/sums, and score consistency.
|
|
54
|
+
JSON is bounded to 2 MiB, 100,000 nodes and depth 64.
|
|
55
|
+
- Model catalogs are advisory. Unknown model IDs remain callable.
|
|
56
|
+
|
|
57
|
+
## OpenAI, Anthropic and Gemini decisions
|
|
58
|
+
|
|
59
|
+
`llmDecisionAdapter({ adapter })` wraps an existing SDK `ModelAdapter`.
|
|
60
|
+
`llmDecisionPlugin({ id, routes, adapter })` installs that bridge into the companion
|
|
61
|
+
runtime. Install only the provider packages you use; the decision package adds no
|
|
62
|
+
dependency on concrete providers. Their existing credentials, gateways, injected
|
|
63
|
+
fetch, request logging and provider-attempt accounting remain in use.
|
|
64
|
+
|
|
65
|
+
OpenAI project examples and live calls use `gpt-6-luna` as the minimum model.
|
|
66
|
+
Select it or a newer model explicitly; do not fall back to older OpenAI models.
|
|
67
|
+
|
|
68
|
+
```ts
|
|
69
|
+
import { createDecisionRuntime, createDecisionTask, choiceQuestion, llmDecisionPlugin }
|
|
70
|
+
from '@alvin0/ai-agent-sdk-decision-adapter'
|
|
71
|
+
import { openAiAdapter } from '@alvin0/ai-agent-sdk-provider-openai'
|
|
72
|
+
import { anthropicAdapter } from '@alvin0/ai-agent-sdk-provider-anthropic'
|
|
73
|
+
import { geminiAdapter } from '@alvin0/ai-agent-sdk-provider-gemini'
|
|
74
|
+
import { envCredential } from '@alvin0/ai-agent-sdk-auth-node/env'
|
|
75
|
+
|
|
76
|
+
const runtime = createDecisionRuntime({ providers: [
|
|
77
|
+
llmDecisionPlugin({ id: 'openai-decisions', routes: ['openai'],
|
|
78
|
+
adapter: openAiAdapter({ apiKey: envCredential('OPENAI_API_KEY') }) }),
|
|
79
|
+
llmDecisionPlugin({ id: 'anthropic-decisions', routes: ['anthropic'],
|
|
80
|
+
adapter: anthropicAdapter({ apiKey: envCredential('ANTHROPIC_API_KEY') }) }),
|
|
81
|
+
llmDecisionPlugin({ id: 'gemini-decisions', routes: ['gemini'],
|
|
82
|
+
adapter: geminiAdapter({ apiKey: envCredential('GEMINI_API_KEY') }) }),
|
|
83
|
+
] })
|
|
84
|
+
try {
|
|
85
|
+
const task = createDecisionTask(runtime.decisionModel({
|
|
86
|
+
provider: 'openai', model: 'gpt-6-luna',
|
|
87
|
+
}), { questions: {
|
|
88
|
+
route: choiceQuestion('Which team should handle this?', {
|
|
89
|
+
billing: 'Invoices and refunds', support: 'Technical issues', other: 'Neither',
|
|
90
|
+
}),
|
|
91
|
+
} })
|
|
92
|
+
const result = await task.evaluate('Please refund my duplicate invoice')
|
|
93
|
+
console.log(result.answers.route.choice)
|
|
94
|
+
// Select provider: 'anthropic' or 'gemini' with that provider's model ID
|
|
95
|
+
// to use the same question/task contract. evaluateBatch works identically.
|
|
96
|
+
} finally { await runtime.close() }
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
The default `outputMode: 'json-schema'` sends a closed JSON Schema through the
|
|
100
|
+
provider's existing structured-output protocol. OpenAI Responses and Chat
|
|
101
|
+
Completions, Anthropic Messages and Gemini Interactions are covered by HTTP fixture
|
|
102
|
+
tests. Choose a model/endpoint supporting that feature. Unsupported options fail
|
|
103
|
+
explicitly; the bridge does not silently switch formats or providers.
|
|
104
|
+
|
|
105
|
+
For endpoints using function calling instead, configure `outputMode: 'tool'`.
|
|
106
|
+
The bridge asks for exactly one `submit_decisions` call and consumes its arguments
|
|
107
|
+
as data. No host tool or agent loop executes. The model must support forced tool
|
|
108
|
+
selection; some models/reasoning configurations reject it. Additional/wrong tool
|
|
109
|
+
calls, truncated responses, invalid JSON, unknown fields/options, and inconsistent
|
|
110
|
+
scores/distributions fail validation. Numeric constraints are checked locally to
|
|
111
|
+
keep the schema portable across provider subsets.
|
|
112
|
+
|
|
113
|
+
`evidence: 'none'` is the default: only choice/score/boolean values are requested.
|
|
114
|
+
Missing probabilities/confidence remain absent. With `evidence: 'model-generated'`,
|
|
115
|
+
Choice/Score request full probability distributions and confidence, and Boolean
|
|
116
|
+
requests `probabilityTrue`. These are model estimates, not native probabilities or
|
|
117
|
+
token logprobs. They are always tagged `model-generated`, cannot spoof provenance,
|
|
118
|
+
and require explicit `allowedSources: ['model-generated']` in a gate. Structured
|
|
119
|
+
output constrains shape; it does not calibrate those estimates.
|
|
120
|
+
|
|
121
|
+
Options also accept `generation: { maxTokens, temperature, topP, reasoningEffort }`
|
|
122
|
+
and `maxResponseBytes` (default 2 MiB, maximum 16 MiB), including reasoning and
|
|
123
|
+
arguments. Sampling/effort support remains model-specific; omission adds no sampling
|
|
124
|
+
defaults. Pass a raw single-attempt adapter, not a retry-wrapped registry/agent
|
|
125
|
+
handle. The decision runtime owns retry/deadline policy and reuses one prepared
|
|
126
|
+
generation across attempts. Direct bridge calls are also deadline-bounded.
|
|
127
|
+
For score questions in evidence mode, the wire schema requests only the
|
|
128
|
+
distribution and confidence. The SDK derives the public fractional score from
|
|
129
|
+
that validated distribution; it does not ask the model to duplicate the arithmetic.
|
|
130
|
+
Extra score fields and malformed distributions are rejected rather than repaired.
|
|
131
|
+
|
|
132
|
+
The SDK generation stream does not expose an authoritative response model ID or
|
|
133
|
+
request ID as public fields. The bridge therefore returns the requested model ID
|
|
134
|
+
and omits `providerRequestId`; provider-attempt telemetry remains with the wrapped
|
|
135
|
+
adapter. Pin model IDs when evaluation reproducibility matters. TypeSafe's native
|
|
136
|
+
adapter continues to preserve its actual response model ID and native evidence.
|
|
137
|
+
|
|
138
|
+
Protocol references: [OpenAI function calling](https://developers.openai.com/api/docs/guides/function-calling),
|
|
139
|
+
[Anthropic tool definitions](https://platform.claude.com/docs/en/agents-and-tools/tool-use/define-tools),
|
|
140
|
+
[Gemini Interactions](https://ai.google.dev/gemini-api/docs/interactions-overview).
|
|
141
|
+
|
|
142
|
+
## Providers and lifecycle
|
|
143
|
+
|
|
144
|
+
Extend `DecisionAdapter`; only `evaluate()` is abstract. One invocation performs
|
|
145
|
+
one physical attempt. Override `resolveModel()` to declare question types and
|
|
146
|
+
provider-specific bounds. Mutable connection configurations must override
|
|
147
|
+
`prepareDecisionCall()` to bind metadata and dispatch to the same generation.
|
|
148
|
+
Its optional fourth argument is the invocation context: forward it during
|
|
149
|
+
connection/credential preparation as well as dispatch. A prepared call belongs to
|
|
150
|
+
one logical request; reuse that request for retries, prepare again for a new input.
|
|
151
|
+
Adapters must honor the request's `signal` and may report physical attempts using
|
|
152
|
+
the supplied SDK `ModelInvocationContext`.
|
|
153
|
+
|
|
154
|
+
`defineDecisionProviderPlugin()` declares versioned route ownership. Register
|
|
155
|
+
multiple plugins with `createDecisionRuntime({ providers })`, or register adapters
|
|
156
|
+
directly with `runtime.registerAdapter(routes, adapter)`. A route cannot have two
|
|
157
|
+
decision adapters; registration is atomic and returns an idempotent disposer.
|
|
158
|
+
|
|
159
|
+
This package supplies a **companion runtime**, not an `AgentRuntime` plugin. Keep
|
|
160
|
+
the decision runtime next to the agent runtime and call its model handles from
|
|
161
|
+
host workflow code or a tool. Pass `ModelInvocationContext` as the second argument
|
|
162
|
+
to `evaluate()` to join existing provider-attempt accounting; this runtime does
|
|
163
|
+
not create an independent observation exporter or fabricate attempt telemetry.
|
|
164
|
+
|
|
165
|
+
The default logical deadline is 30 seconds, including preparation, attempts and
|
|
166
|
+
backoff. Override with runtime `timeoutMs` or per-call `timeoutMs`. SDK retry policy
|
|
167
|
+
and backoff helpers are reused; the adapter's route retry policy takes precedence
|
|
168
|
+
over the runtime default when explicitly configured (TypeSafe inherits the runtime
|
|
169
|
+
policy when its `retryPolicy` is omitted). A single prepared call is reused across
|
|
170
|
+
retries. The runtime owns the overall deadline; the LLM bridge does not add an
|
|
171
|
+
independent 30-second deadline inside a runtime call. Direct LLM calls retain a
|
|
172
|
+
30-second default. Transport request limits may still bound a physical attempt.
|
|
173
|
+
`close()` aborts active work, runs plugin cleanup, and rejects subsequent calls.
|
|
174
|
+
Noncooperative extension promises are detached after cancellation; their late
|
|
175
|
+
rejections are observed, but external side effects cannot be undone.
|
|
176
|
+
|
|
177
|
+
### Connecting to a core agent
|
|
178
|
+
|
|
179
|
+
The current `AgentRuntime.providers` accepts generation and embedding plugins,
|
|
180
|
+
not decision plugins, and `AgentRuntime` has no `decisionModel()` method. A decision
|
|
181
|
+
task can nevertheless run inside a core tool through the public API:
|
|
182
|
+
|
|
183
|
+
```ts
|
|
184
|
+
import { defineTool } from '@alvin0/ai-agent-sdk-core'
|
|
185
|
+
|
|
186
|
+
// `task` is a configured decision task with a `route` choice question.
|
|
187
|
+
const classifyTicket = defineTool<{ message: string }>({
|
|
188
|
+
name: 'classify_ticket', description: 'Classify a support ticket',
|
|
189
|
+
parameters: {
|
|
190
|
+
type: 'object', properties: { message: { type: 'string' } },
|
|
191
|
+
required: ['message'], additionalProperties: false,
|
|
192
|
+
},
|
|
193
|
+
parse(raw) {
|
|
194
|
+
if (raw === null || typeof raw !== 'object' || !('message' in raw)
|
|
195
|
+
|| typeof raw.message !== 'string') throw new Error('message required')
|
|
196
|
+
return { message: raw.message }
|
|
197
|
+
},
|
|
198
|
+
async execute(args, ctx) {
|
|
199
|
+
const result = await task.evaluate({ message: args.message },
|
|
200
|
+
{ signal: ctx.signal }, { ...(ctx.logger ? { logger: ctx.logger } : {}) })
|
|
201
|
+
return { route: result.answers.route.choice, model: result.model }
|
|
202
|
+
},
|
|
203
|
+
})
|
|
204
|
+
// Pass tools: [classifyTicket] when binding a core agent/session.
|
|
205
|
+
```
|
|
206
|
+
|
|
207
|
+
Return the selected JSON fields to the agent, and always forward `ctx.signal`.
|
|
208
|
+
Closing core aborts a decision called by an active tool when that signal is
|
|
209
|
+
forwarded. Core does not close the independently-created decision runtime; the
|
|
210
|
+
host must close both owners, including decisions called outside agent runs.
|
|
211
|
+
Tool signals are scoped to one invocation and are aborted during cleanup, so do
|
|
212
|
+
not keep them for later background work.
|
|
213
|
+
|
|
214
|
+
`ToolRunContext` exposes cancellation, logging and run position; it does not
|
|
215
|
+
expose a complete `ModelInvocationContext` or provider-attempt accounting handle.
|
|
216
|
+
Forwarding its logger alone does not merge nested decision usage into the core
|
|
217
|
+
agent's token totals, budgets or model-call reports. A host-owned invocation
|
|
218
|
+
context can be passed explicitly; full core-managed admission, lifecycle,
|
|
219
|
+
catalogs and accounting would require a native decision capability in core.
|
|
220
|
+
|
|
221
|
+
The public integration tests cover tool output reaching the next agent step,
|
|
222
|
+
core-close cancellation, independent companion lifecycle, and rejection of
|
|
223
|
+
decision plugins in the core provider list.
|
|
224
|
+
|
|
225
|
+
Run `pnpm --filter @alvin0/ai-agent-sdk-decision-adapter test` for offline contract
|
|
226
|
+
tests and `test:pack` for installed-tarball verification.
|
|
227
|
+
For live acceptance with workspace `.env` keys, run `pnpm human:decision:all` after
|
|
228
|
+
building. It covers TypeSafe and OpenAI JSON Schema/tool mode, English/Vietnamese,
|
|
229
|
+
batches, model-generated evidence and a core agent tool. Unconfigured providers
|
|
230
|
+
are skipped. OpenAI defaults to `gpt-6-luna`. `OPENAI_DECISION_MODEL` (or
|
|
231
|
+
`OPENAI_MODEL`) may select it or a newer model; `TYPESAFE_MODEL` pins the TypeSafe
|
|
232
|
+
model. The runner emits safe status/usage/latency diagnostics without keys or raw
|
|
233
|
+
errors.
|
|
234
|
+
|
|
235
|
+
## Reusable tasks and batches
|
|
236
|
+
|
|
237
|
+
`createDecisionTask(model, { questions, timeoutMs, context })` binds a detached,
|
|
238
|
+
frozen rubric to a model handle. `task.evaluate(state, { signal, timeoutMs }, context)`
|
|
239
|
+
overrides call options without repeating questions or connection configuration.
|
|
240
|
+
SDK-created immutable snapshots are reused; arbitrary frozen caller inputs still
|
|
241
|
+
undergo validation. Task batches share the captured rubric across queued states.
|
|
242
|
+
The LLM bridge and TypeSafe compile each rubric once per adapter/rubric identity,
|
|
243
|
+
and reuse the captured request through retries. No result cache or cross-call
|
|
244
|
+
credential/connection cache is introduced. Prepared calls reject requests targeting
|
|
245
|
+
a different provider/model before dispatch. Direct LLM prepared evaluations also
|
|
246
|
+
honor `timeoutMs` (30 seconds by default) and close stalled iterators without
|
|
247
|
+
waiting for their teardown. JSON arrays must contain only their own enumerable
|
|
248
|
+
indexed data properties; getters, custom iterators, symbols and auxiliary
|
|
249
|
+
properties are rejected without running caller hooks. LLM prompts put the fixed questions
|
|
250
|
+
before changing state; evidence instructions appear only in evidence mode.
|
|
251
|
+
Use a separate task per model/rubric combination; changing providers requires no
|
|
252
|
+
change to the task's question contract.
|
|
253
|
+
|
|
254
|
+
`task.evaluateBatch(states, { concurrency, signal, timeoutMs, context })` evaluates
|
|
255
|
+
independent states with the same rubric. `evaluateDecisionBatch(model, inputs, options)`
|
|
256
|
+
also accepts a different question set per input. Both preserve input order and
|
|
257
|
+
return `{ status: 'fulfilled', value }` or `{ status: 'rejected', reason }` per item.
|
|
258
|
+
They use at most four concurrent logical calls by default (configurable 1–256).
|
|
259
|
+
This is client-side scheduling, not a provider batch endpoint. Retries occupy the
|
|
260
|
+
same concurrency slot. A batch accepts at most 1,024 inputs, snapshots queued
|
|
261
|
+
payloads before dispatch, and does not automatically chunk or truncate data.
|
|
262
|
+
|
|
263
|
+
The batch deadline defaults to 30 seconds and includes queue time. Batch `timeoutMs`
|
|
264
|
+
sets that overall deadline; task `timeoutMs` or input `timeoutMs` still governs each
|
|
265
|
+
dispatched call, including custom handles that ignore deadlines. Item deadlines
|
|
266
|
+
start on dispatch, independently of the overall deadline that includes queue time.
|
|
267
|
+
Batch cancellation preserves caller `ModelError` reasons (including `TIMEOUT`)
|
|
268
|
+
and signal composition supports high concurrency without accumulating listeners
|
|
269
|
+
on the shared runtime/batch signal. Batch cancellation/deadline rejects the operation and stops queued
|
|
270
|
+
dispatches; individual item failures/cancellation remain partial results. Invalid
|
|
271
|
+
input/configuration rejects the batch before dispatch. Empty batches return `[]`.
|
|
272
|
+
For large datasets, submit application-managed chunks with explicit deadlines.
|
|
273
|
+
Concurrency limits active logical calls, not the memory used by queued snapshots;
|
|
274
|
+
choose chunk sizes based on payload size as well as item count.
|
|
275
|
+
|
|
276
|
+
## Evidence gates
|
|
277
|
+
|
|
278
|
+
`gateChoice(answer, { minConfidence, minProbability, minMargin, allowedSources })`
|
|
279
|
+
requires at least one explicit threshold and accepts only when every configured
|
|
280
|
+
threshold passes. `minMargin` compares the selected probability to the runner-up.
|
|
281
|
+
`gateBoolean(answer, { falseMax, trueMin, allowedSources })` accepts false below or
|
|
282
|
+
at `falseMax`, true above or at `trueMin`, and abstains between them. Both return
|
|
283
|
+
`{ status: 'accepted', value }` or `{ status: 'abstained', reason }`.
|
|
284
|
+
|
|
285
|
+
Gates consume validated answers and default to `allowedSources: ['provider']`.
|
|
286
|
+
Missing required evidence or disallowed provenance causes abstention. These pure
|
|
287
|
+
helpers do not call a fallback model, run tools, derive confidence, or authorize
|
|
288
|
+
actions. Calibrate thresholds against labeled examples for each model, question
|
|
289
|
+
and domain. A provider confidence statistic is not an empirical accuracy estimate.
|
package/USE_CASES.md
ADDED
|
@@ -0,0 +1,217 @@
|
|
|
1
|
+
# Use cases and setup
|
|
2
|
+
|
|
3
|
+
The kit separates three concerns: the provider handles transport and model
|
|
4
|
+
capabilities; a task binds a model to reusable questions; application code owns
|
|
5
|
+
thresholds, branches, ranking and actions. A decision model selects from a closed
|
|
6
|
+
answer space. Generated prose, open-ended tool arguments and arbitrary extraction
|
|
7
|
+
remain generation tasks.
|
|
8
|
+
|
|
9
|
+
## Choosing the decision shape
|
|
10
|
+
|
|
11
|
+
| Use case | Question setup | Host workflow | Main pitfall |
|
|
12
|
+
| --- | --- | --- | --- |
|
|
13
|
+
| Intent, tool or model routing | Choice over known routes, including `other` | Gate the selected route; select a handler/model | A forced winner is not proof the request fits any route |
|
|
14
|
+
| Support triage / semantic features | Choice + independent booleans + ordinal scores in one request | Combine relevant answers in code | Questions do not consume each other's answers |
|
|
15
|
+
| Input/output/tool verification | One boolean per failure mode, with concrete true/false criteria | Block, review or continue using a three-way gate | An uncertain negative does not prove safety |
|
|
16
|
+
| RAG reranking / entity matching | Same relevance/match question on each query-candidate pair | Bounded batch, then filter and sort successful candidates | Choice probabilities are relative to the candidate set |
|
|
17
|
+
| Severity / lead prioritization | Ordered score rubric, or several independent indicators | Apply application weights and thresholds | A fractional expected score is not an integer label or a calibrated business value |
|
|
18
|
+
| Known-value extraction | Choice among parser-generated candidates plus `none` | Return the selected candidate's original value | A decision primitive cannot invent a missing address/date/argument |
|
|
19
|
+
| Hierarchical classification / cascades | Choice at each branch or a verification task after selection | Separate calls with explicit updated state and a depth/budget limit | A greedy early mistake propagates; multiplying evidence does not guarantee a calibrated path probability |
|
|
20
|
+
|
|
21
|
+
These patterns follow TypeSafe's [use-case map](https://docs.typesafe.ai/concepts/use-case-map),
|
|
22
|
+
[fan-out](https://docs.typesafe.ai/patterns/fan-out),
|
|
23
|
+
[reranking](https://docs.typesafe.ai/cookbooks/rerank_typesafe), and
|
|
24
|
+
[hierarchical classification](https://docs.typesafe.ai/cookbooks/hierarchical_classification)
|
|
25
|
+
examples. They are workflow patterns rather than provider-specific task types.
|
|
26
|
+
|
|
27
|
+
## Setup once, reuse across consumers
|
|
28
|
+
|
|
29
|
+
For a service, create one runtime during startup, inject task handles into request
|
|
30
|
+
handlers/tools, and close the runtime during shutdown. For a CLI/job, use
|
|
31
|
+
`try/finally`. Connection configuration is separate from the rubric so consumers
|
|
32
|
+
do not receive credentials. Pass request-scoped invocation context explicitly
|
|
33
|
+
instead of storing it on a singleton task.
|
|
34
|
+
|
|
35
|
+
```ts
|
|
36
|
+
import {
|
|
37
|
+
createDecisionRuntime, createDecisionTask, choiceQuestion, booleanQuestion,
|
|
38
|
+
scoreQuestion, gateChoice, gateBoolean,
|
|
39
|
+
} from '@alvin0/ai-agent-sdk-decision-adapter'
|
|
40
|
+
import { typesafePlugin } from '@alvin0/ai-agent-sdk-provider-typesafe'
|
|
41
|
+
import { envCredential } from '@alvin0/ai-agent-sdk-auth-node/env'
|
|
42
|
+
|
|
43
|
+
const runtime = createDecisionRuntime({
|
|
44
|
+
providers: [typesafePlugin({ apiKey: envCredential('TYPESAFE_API_KEY') })],
|
|
45
|
+
timeoutMs: 5_000,
|
|
46
|
+
retryPolicy: { mode: 'normal', maxRetries: 1 },
|
|
47
|
+
})
|
|
48
|
+
const model = runtime.decisionModel({ provider: 'typesafe', model: 'jev-1.13.0' })
|
|
49
|
+
const triage = createDecisionTask(model, {
|
|
50
|
+
timeoutMs: 3_000,
|
|
51
|
+
questions: {
|
|
52
|
+
route: choiceQuestion('Which team should handle this ticket?', {
|
|
53
|
+
billing: 'Invoices, charges and refunds',
|
|
54
|
+
support: 'Product failures and technical issues',
|
|
55
|
+
other: 'Outside the listed categories or insufficient information',
|
|
56
|
+
}),
|
|
57
|
+
severity: scoreQuestion('Impact of the technical issue, if present', [
|
|
58
|
+
'Cosmetic', 'Degraded with a workaround', 'Blocked without a workaround',
|
|
59
|
+
]),
|
|
60
|
+
refund: booleanQuestion('Is a refund explicitly requested?', {
|
|
61
|
+
true: 'An explicit request for money back or a credit',
|
|
62
|
+
false: 'No explicit refund request, including merely asking about prices',
|
|
63
|
+
}),
|
|
64
|
+
},
|
|
65
|
+
})
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
Pin a tested model version for stable evaluation/calibration. An alias such as
|
|
69
|
+
`jev-latest` is convenient for exploration; keep `result.model` in evaluation
|
|
70
|
+
records so an alias change is visible. Models need no catalog allowlist. Other
|
|
71
|
+
providers can bind the same questions through their own decision plugins, subject
|
|
72
|
+
to their declared capabilities. Do not assume identical limits or probability
|
|
73
|
+
semantics across providers.
|
|
74
|
+
|
|
75
|
+
### Routing and triage
|
|
76
|
+
|
|
77
|
+
```ts
|
|
78
|
+
const result = await triage.evaluate({ message: 'I was charged twice; please refund.' })
|
|
79
|
+
const route = gateChoice(result.answers.route, {
|
|
80
|
+
minProbability: 0.85, minMargin: 0.2,
|
|
81
|
+
})
|
|
82
|
+
const refund = gateBoolean(result.answers.refund, { falseMax: 0.2, trueMin: 0.8 })
|
|
83
|
+
// Example thresholds only; measure them against your own labeled tickets.
|
|
84
|
+
if (route.status === 'abstained') {
|
|
85
|
+
// Queue for review or ask a separate model; retain the original result.
|
|
86
|
+
} else if (route.value === 'billing') {
|
|
87
|
+
// Attach refund.status/value to the billing queue, rather than issuing a refund.
|
|
88
|
+
} else if (route.value === 'support') {
|
|
89
|
+
// Read severity here; the billing path can ignore that speculative answer.
|
|
90
|
+
}
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
Multiple questions in one request all see the same state. This supports multi-label
|
|
94
|
+
tagging through independent booleans; a single Choice returns one mutually exclusive
|
|
95
|
+
winner. Questions whose rubric actually depends on an earlier answer require a
|
|
96
|
+
second call with that answer included in the next state.
|
|
97
|
+
|
|
98
|
+
### Guardrails and verification
|
|
99
|
+
|
|
100
|
+
Use a reusable boolean task for each input/output/tool check, passing the artifact,
|
|
101
|
+
relevant policy and evidence in the state. Name the question so the polarity is
|
|
102
|
+
clear, for example `violatesPolicy`, rather than an ambiguous `safe`.
|
|
103
|
+
|
|
104
|
+
```ts
|
|
105
|
+
const verify = createDecisionTask(model, { questions: {
|
|
106
|
+
violatesPolicy: booleanQuestion('Does the proposed response violate the supplied policy?', {
|
|
107
|
+
true: 'At least one concrete policy violation is supported by the response',
|
|
108
|
+
false: 'The proposed response complies with every applicable supplied rule',
|
|
109
|
+
}),
|
|
110
|
+
} })
|
|
111
|
+
const check = await verify.evaluate({ policy: 'Do not disclose internal access tokens.', response: 'Hello!' })
|
|
112
|
+
const verdict = gateBoolean(check.answers.violatesPolicy, { falseMax: 0.1, trueMin: 0.8 })
|
|
113
|
+
// accepted false: continue; accepted true: block; abstained: review/fallback.
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
Transport/auth failures throw; they do not become a semantic negative. The host
|
|
117
|
+
sets failure behavior. Decision checks supplement deterministic permission and
|
|
118
|
+
schema checks; the gate itself executes no action. For citations, supply both the
|
|
119
|
+
claim and source passage. For tool selection, validate tool arguments separately.
|
|
120
|
+
|
|
121
|
+
### Rerank a shortlist or match entities
|
|
122
|
+
|
|
123
|
+
For a runnable example combining access/date filtering, per-facet document
|
|
124
|
+
selection and OpenAI answers with exact source quotes, see the
|
|
125
|
+
[document-selection sample](../../samples/decision-document-selection/README.md).
|
|
126
|
+
It also handles historical policies and abstains when evidence is incomplete.
|
|
127
|
+
|
|
128
|
+
Use one state per candidate so relevance scores do not depend on which other
|
|
129
|
+
candidates happen to share a request. Keep candidate IDs in the host array.
|
|
130
|
+
|
|
131
|
+
```ts
|
|
132
|
+
const relevance = createDecisionTask(model, { questions: {
|
|
133
|
+
relevant: booleanQuestion('Does this passage directly answer the query?', {
|
|
134
|
+
true: 'Contains information sufficient to answer the query',
|
|
135
|
+
false: 'Only shares a topic or lacks the needed information',
|
|
136
|
+
}),
|
|
137
|
+
} })
|
|
138
|
+
const candidates = [
|
|
139
|
+
{ id: 'a', text: 'Refunds are processed in five business days.' },
|
|
140
|
+
{ id: 'b', text: 'Our logo is blue.' },
|
|
141
|
+
]
|
|
142
|
+
const rows = await relevance.evaluateBatch(candidates.map(candidate => ({
|
|
143
|
+
query: 'When will my refund arrive?', passage: candidate.text,
|
|
144
|
+
})), { concurrency: 4, timeoutMs: 15_000 })
|
|
145
|
+
const ranked = rows.flatMap((row, index) => {
|
|
146
|
+
if (row.status !== 'fulfilled') return [] // Record failures separately; do not score them as zero.
|
|
147
|
+
const answer = row.value.answers.relevant
|
|
148
|
+
if (answer.probabilitySource !== 'provider' || answer.probabilityTrue === undefined) return []
|
|
149
|
+
return [{ candidate: candidates[index]!, score: answer.probabilityTrue, result: row.value }]
|
|
150
|
+
}).sort((a, b) => b.score - a.score)
|
|
151
|
+
```
|
|
152
|
+
|
|
153
|
+
Retrieval produces the shortlist first; reranking cannot recover missing candidates.
|
|
154
|
+
Use this pattern for record linkage by supplying two candidate records and a fixed
|
|
155
|
+
match rubric. Prefer a labeled evaluation set to interpreting a raw probability
|
|
156
|
+
as a guaranteed match rate. The batch preserves order, partial failures and actual
|
|
157
|
+
model IDs. Configure its overall deadline for the entire queue, not just one call.
|
|
158
|
+
|
|
159
|
+
### Staged decisions and model fallback
|
|
160
|
+
|
|
161
|
+
The host can first call a routing task, gate its evidence, then call a specialized
|
|
162
|
+
task with `{ originalState, selectedRoute }`. For an uncertain result, explicitly
|
|
163
|
+
invoke another configured decision handle or an existing generation model. Treat
|
|
164
|
+
semantic abstention separately from network failure. Never silently relabel an
|
|
165
|
+
auth/configuration failure as uncertainty. Keep each stage's result for audit.
|
|
166
|
+
|
|
167
|
+
For a taxonomy, bound depth, model calls and time; optional beam search explores
|
|
168
|
+
several probable branches rather than only the first winner. For candidate
|
|
169
|
+
extraction, let deterministic parsing produce candidates, select among their IDs,
|
|
170
|
+
and verify the chosen value in a second stage. Include `none`/`other` where absence
|
|
171
|
+
is valid. Split independent work into batches; keep dependencies sequential.
|
|
172
|
+
|
|
173
|
+
## Provider and environment choices
|
|
174
|
+
|
|
175
|
+
| Consumer | Setup |
|
|
176
|
+
| --- | --- |
|
|
177
|
+
| Node service / agent tool | Lazy `envCredential`, shared runtime/tasks, per-request signal/context |
|
|
178
|
+
| Browser app | Server endpoint holding the key; inject a server-side task behind it |
|
|
179
|
+
| Worker / edge runtime | Inject a key from the host's secret binding or a `CredentialSource`; use Web fetch |
|
|
180
|
+
| Local/self-hosted API | Adapter with `baseUrl`, injected transport, explicit HTTP opt-in for local development |
|
|
181
|
+
| Multiple accounts/endpoints | Multiple TypeSafe plugins with distinct `id` and `routes`; bind a task to each route |
|
|
182
|
+
| Unit tests / offline execution | Register a deterministic `DecisionAdapter` or inject mocked fetch; keep the same task/gate code |
|
|
183
|
+
|
|
184
|
+
Credential resolution follows the underlying adapter: TypeSafe resolves per physical
|
|
185
|
+
attempt; prepared LLM adapters may capture credentials once per logical call.
|
|
186
|
+
An explicitly configured provider retry policy overrides the runtime default;
|
|
187
|
+
TypeSafe inherits the runtime policy when omitted. Choose one retry owner to avoid compounded attempts. Cancellation
|
|
188
|
+
bounds scheduling and waiting; it does not undo provider processing or host actions.
|
|
189
|
+
No batch cache, automatic fallback, model-specific calibration, business side effects,
|
|
190
|
+
or streaming dataset ingestion is implied by this package.
|
|
191
|
+
|
|
192
|
+
See TypeSafe's [confidence explanation](https://docs.typesafe.ai/confidence) for
|
|
193
|
+
the distribution statistic behind Choice/Score confidence. A Boolean's probability
|
|
194
|
+
near 0.5 is uncertain; it has no native separate confidence. The kit preserves raw
|
|
195
|
+
evidence and requires explicit gates rather than fabricating a shared metric.
|
|
196
|
+
|
|
197
|
+
## Use an LLM for the same tasks
|
|
198
|
+
|
|
199
|
+
Wrap `openAiAdapter`, `anthropicAdapter`, `geminiAdapter` or another SDK model
|
|
200
|
+
adapter in `llmDecisionPlugin`. The [LLM setup example](./README.md#openai-anthropic-and-gemini-decisions)
|
|
201
|
+
registers all three. Keep the rubric and batch/workflow code; change the target
|
|
202
|
+
provider/model to select the implementation. Different providers can coexist with
|
|
203
|
+
TypeSafe in the same decision runtime.
|
|
204
|
+
|
|
205
|
+
Use JSON Schema by default, or select the tool mode for a compatible endpoint.
|
|
206
|
+
Generation must finish successfully before its answer is accepted; a partially
|
|
207
|
+
valid answer cut off by the token limit is rejected. No format repair or automatic
|
|
208
|
+
fallback is hidden in the adapter. Host cascades can explicitly catch failures or
|
|
209
|
+
handle abstention and invoke a different configured task.
|
|
210
|
+
|
|
211
|
+
LLM decisions default to values without confidence. For a routing policy that
|
|
212
|
+
needs raw evidence, prefer a provider that supplies it, or explicitly opt into
|
|
213
|
+
`evidence: 'model-generated'` and validate its behavior on your dataset. Existing
|
|
214
|
+
gates abstain on absent evidence and disallow these self-reported estimates by
|
|
215
|
+
default. Configuring structured output does not turn an LLM into a calibrated
|
|
216
|
+
native decision model. A batch may span an alias update; pinned IDs and per-stage
|
|
217
|
+
records help evaluation, while response model IDs are unavailable on this bridge.
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
import { MODEL_ERROR_CODES, ModelError } from "@alvin0/ai-agent-sdk-core";
|
|
2
|
+
|
|
3
|
+
//#region src/async.ts
|
|
4
|
+
function throwIfAborted(signal) {
|
|
5
|
+
if (signal.aborted) throw signal.reason instanceof ModelError ? signal.reason : new ModelError("Decision call aborted", MODEL_ERROR_CODES.ABORTED);
|
|
6
|
+
}
|
|
7
|
+
/** Settles promptly even if an extension ignores cancellation; observes its late rejection. */
|
|
8
|
+
function abortable(work, signal) {
|
|
9
|
+
return new Promise((resolve, reject) => {
|
|
10
|
+
const abort = () => {
|
|
11
|
+
cleanup();
|
|
12
|
+
try {
|
|
13
|
+
throwIfAborted(signal);
|
|
14
|
+
} catch (error) {
|
|
15
|
+
reject(error);
|
|
16
|
+
}
|
|
17
|
+
};
|
|
18
|
+
const cleanup = () => signal.removeEventListener("abort", abort);
|
|
19
|
+
work.then((value) => {
|
|
20
|
+
cleanup();
|
|
21
|
+
resolve(value);
|
|
22
|
+
}, (error) => {
|
|
23
|
+
cleanup();
|
|
24
|
+
reject(error);
|
|
25
|
+
});
|
|
26
|
+
signal.addEventListener("abort", abort, { once: true });
|
|
27
|
+
if (signal.aborted) abort();
|
|
28
|
+
});
|
|
29
|
+
}
|
|
30
|
+
function waitDecisionDelay(ms, signal) {
|
|
31
|
+
return new Promise((resolve, reject) => {
|
|
32
|
+
const abort = () => {
|
|
33
|
+
clearTimeout(timer);
|
|
34
|
+
cleanup();
|
|
35
|
+
try {
|
|
36
|
+
throwIfAborted(signal);
|
|
37
|
+
} catch (error) {
|
|
38
|
+
reject(error);
|
|
39
|
+
}
|
|
40
|
+
};
|
|
41
|
+
const cleanup = () => signal.removeEventListener("abort", abort);
|
|
42
|
+
const timer = setTimeout(() => {
|
|
43
|
+
cleanup();
|
|
44
|
+
resolve();
|
|
45
|
+
}, ms);
|
|
46
|
+
signal.addEventListener("abort", abort, { once: true });
|
|
47
|
+
if (signal.aborted) abort();
|
|
48
|
+
});
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
//#endregion
|
|
52
|
+
export { throwIfAborted as n, waitDecisionDelay as r, abortable as t };
|
|
53
|
+
//# sourceMappingURL=async-BPqlUFnU.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"async-BPqlUFnU.js","names":[],"sources":["../src/async.ts"],"sourcesContent":["import { ModelError, MODEL_ERROR_CODES } from '@alvin0/ai-agent-sdk-core'\n\nexport function throwIfAborted(signal: AbortSignal): void {\n if (signal.aborted) throw signal.reason instanceof ModelError ? signal.reason : new ModelError('Decision call aborted', MODEL_ERROR_CODES.ABORTED)\n}\n/** Settles promptly even if an extension ignores cancellation; observes its late rejection. */\nexport function abortable<T>(work: Promise<T>, signal: AbortSignal): Promise<T> {\n return new Promise<T>((resolve, reject) => {\n const abort = () => { cleanup(); try { throwIfAborted(signal) } catch (error) { reject(error) } }\n const cleanup = () => signal.removeEventListener('abort', abort)\n work.then(value => { cleanup(); resolve(value) }, error => { cleanup(); reject(error) })\n signal.addEventListener('abort', abort, { once: true })\n if (signal.aborted) abort()\n })\n}\nexport function waitDecisionDelay(ms: number, signal: AbortSignal): Promise<void> {\n return new Promise((resolve, reject) => {\n const abort = () => { clearTimeout(timer); cleanup(); try { throwIfAborted(signal) } catch (error) { reject(error) } }\n const cleanup = () => signal.removeEventListener('abort', abort)\n const timer = setTimeout(() => { cleanup(); resolve() }, ms)\n signal.addEventListener('abort', abort, { once: true })\n if (signal.aborted) abort()\n })\n}\n"],"mappings":";;;AAEA,SAAgB,eAAe,QAA2B;CACxD,IAAI,OAAO,SAAS,MAAM,OAAO,kBAAkB,aAAa,OAAO,SAAS,IAAI,WAAW,yBAAyB,kBAAkB,OAAO;AACnJ;;AAEA,SAAgB,UAAa,MAAkB,QAAiC;CAC9E,OAAO,IAAI,SAAY,SAAS,WAAW;EACzC,MAAM,cAAc;GAAE,QAAQ;GAAG,IAAI;IAAE,eAAe,MAAM;GAAE,SAAS,OAAO;IAAE,OAAO,KAAK;GAAE;EAAE;EAChG,MAAM,gBAAgB,OAAO,oBAAoB,SAAS,KAAK;EAC/D,KAAK,MAAK,UAAS;GAAE,QAAQ;GAAG,QAAQ,KAAK;EAAE,IAAG,UAAS;GAAE,QAAQ;GAAG,OAAO,KAAK;EAAE,CAAC;EACvF,OAAO,iBAAiB,SAAS,OAAO,EAAE,MAAM,KAAK,CAAC;EACtD,IAAI,OAAO,SAAS,MAAM;CAC5B,CAAC;AACH;AACA,SAAgB,kBAAkB,IAAY,QAAoC;CAChF,OAAO,IAAI,SAAS,SAAS,WAAW;EACtC,MAAM,cAAc;GAAE,aAAa,KAAK;GAAG,QAAQ;GAAG,IAAI;IAAE,eAAe,MAAM;GAAE,SAAS,OAAO;IAAE,OAAO,KAAK;GAAE;EAAE;EACrH,MAAM,gBAAgB,OAAO,oBAAoB,SAAS,KAAK;EAC/D,MAAM,QAAQ,iBAAiB;GAAE,QAAQ;GAAG,QAAQ;EAAE,GAAG,EAAE;EAC3D,OAAO,iBAAiB,SAAS,OAAO,EAAE,MAAM,KAAK,CAAC;EACtD,IAAI,OAAO,SAAS,MAAM;CAC5B,CAAC;AACH"}
|