@alvin0/ai-agent-sdk-decision-adapter 0.0.0-stage → 0.1.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 alvin0 (chaulamdinhai) <chaulamdinhai@gmail.com>
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
package/README.md CHANGED
@@ -1,3 +1,289 @@
1
- # Temporary Holding Version
1
+ # Decision adapter
2
2
 
3
- This version is a temporary placeholder for this package. An operational version to replace this has been submitted for review and is awaiting a staged release.
3
+ Runtime: **Universal** (Node, browser, and Web-standard workers).
4
+
5
+ ```sh
6
+ pnpm add @alvin0/ai-agent-sdk-core @alvin0/ai-agent-sdk-decision-adapter
7
+ ```
8
+
9
+ Provider-neutral typed decisions, independently of chat generation and embeddings.
10
+ Use `choiceQuestion`, `scoreQuestion`, and `booleanQuestion` to define a closed
11
+ answer space. The return type preserves question IDs and literal choice options.
12
+
13
+ See [Use cases and setup](./USE_CASES.md) for routing, triage, guardrails, RAG
14
+ reranking, extraction, and staged workflows, including their tradeoffs.
15
+
16
+ ```ts
17
+ import {
18
+ createDecisionRuntime, choiceQuestion,
19
+ } from '@alvin0/ai-agent-sdk-decision-adapter'
20
+ import { typesafePlugin } from '@alvin0/ai-agent-sdk-provider-typesafe'
21
+
22
+ const runtime = createDecisionRuntime({
23
+ providers: [typesafePlugin({ apiKey: 'your-server-side-key' })],
24
+ })
25
+ try {
26
+ const model = runtime.decisionModel({ provider: 'typesafe', model: 'jev-latest' })
27
+ const result = await model.evaluate({
28
+ state: { message: 'Please refund my invoice' },
29
+ questions: {
30
+ department: choiceQuestion('Which department should handle this?', {
31
+ billing: 'Invoices and refunds', support: 'Technical issues',
32
+ }),
33
+ },
34
+ })
35
+ // Typed as 'billing' | 'support'.
36
+ console.log(result.answers.department.choice)
37
+ } finally {
38
+ await runtime.close()
39
+ }
40
+ ```
41
+
42
+ ## Contract
43
+
44
+ - `choice`: selected option; optional full probability distribution.
45
+ - `score`: expected zero-based ordinal level; fractional scores are valid.
46
+ - `boolean`: boolean value; optional probability of true. The TypeSafe provider
47
+ uses `probabilityTrue >= 0.5` for `value`; application thresholds can use the raw
48
+ probability instead.
49
+ - Evidence carries `probabilitySource`: `provider`, `token-logprobs`, or
50
+ `model-generated`. The tag identifies provenance, not a calibration guarantee.
51
+ Missing evidence/usage stays absent. Confidence is not inferred.
52
+ - JSON is detached and frozen before async preparation. Results are checked for
53
+ IDs, types, allowed choices, probability bounds/sums, and score consistency.
54
+ JSON is bounded to 2 MiB, 100,000 nodes and depth 64.
55
+ - Model catalogs are advisory. Unknown model IDs remain callable.
56
+
57
+ ## OpenAI, Anthropic and Gemini decisions
58
+
59
+ `llmDecisionAdapter({ adapter })` wraps an existing SDK `ModelAdapter`.
60
+ `llmDecisionPlugin({ id, routes, adapter })` installs that bridge into the companion
61
+ runtime. Install only the provider packages you use; the decision package adds no
62
+ dependency on concrete providers. Their existing credentials, gateways, injected
63
+ fetch, request logging and provider-attempt accounting remain in use.
64
+
65
+ OpenAI project examples and live calls use `gpt-6-luna` as the minimum model.
66
+ Select it or a newer model explicitly; do not fall back to older OpenAI models.
67
+
68
+ ```ts
69
+ import { createDecisionRuntime, createDecisionTask, choiceQuestion, llmDecisionPlugin }
70
+ from '@alvin0/ai-agent-sdk-decision-adapter'
71
+ import { openAiAdapter } from '@alvin0/ai-agent-sdk-provider-openai'
72
+ import { anthropicAdapter } from '@alvin0/ai-agent-sdk-provider-anthropic'
73
+ import { geminiAdapter } from '@alvin0/ai-agent-sdk-provider-gemini'
74
+ import { envCredential } from '@alvin0/ai-agent-sdk-auth-node/env'
75
+
76
+ const runtime = createDecisionRuntime({ providers: [
77
+ llmDecisionPlugin({ id: 'openai-decisions', routes: ['openai'],
78
+ adapter: openAiAdapter({ apiKey: envCredential('OPENAI_API_KEY') }) }),
79
+ llmDecisionPlugin({ id: 'anthropic-decisions', routes: ['anthropic'],
80
+ adapter: anthropicAdapter({ apiKey: envCredential('ANTHROPIC_API_KEY') }) }),
81
+ llmDecisionPlugin({ id: 'gemini-decisions', routes: ['gemini'],
82
+ adapter: geminiAdapter({ apiKey: envCredential('GEMINI_API_KEY') }) }),
83
+ ] })
84
+ try {
85
+ const task = createDecisionTask(runtime.decisionModel({
86
+ provider: 'openai', model: 'gpt-6-luna',
87
+ }), { questions: {
88
+ route: choiceQuestion('Which team should handle this?', {
89
+ billing: 'Invoices and refunds', support: 'Technical issues', other: 'Neither',
90
+ }),
91
+ } })
92
+ const result = await task.evaluate('Please refund my duplicate invoice')
93
+ console.log(result.answers.route.choice)
94
+ // Select provider: 'anthropic' or 'gemini' with that provider's model ID
95
+ // to use the same question/task contract. evaluateBatch works identically.
96
+ } finally { await runtime.close() }
97
+ ```
98
+
99
+ The default `outputMode: 'json-schema'` sends a closed JSON Schema through the
100
+ provider's existing structured-output protocol. OpenAI Responses and Chat
101
+ Completions, Anthropic Messages and Gemini Interactions are covered by HTTP fixture
102
+ tests. Choose a model/endpoint supporting that feature. Unsupported options fail
103
+ explicitly; the bridge does not silently switch formats or providers.
104
+
105
+ For endpoints using function calling instead, configure `outputMode: 'tool'`.
106
+ The bridge asks for exactly one `submit_decisions` call and consumes its arguments
107
+ as data. No host tool or agent loop executes. The model must support forced tool
108
+ selection; some models/reasoning configurations reject it. Additional/wrong tool
109
+ calls, truncated responses, invalid JSON, unknown fields/options, and inconsistent
110
+ scores/distributions fail validation. Numeric constraints are checked locally to
111
+ keep the schema portable across provider subsets.
112
+
113
+ `evidence: 'none'` is the default: only choice/score/boolean values are requested.
114
+ Missing probabilities/confidence remain absent. With `evidence: 'model-generated'`,
115
+ Choice/Score request full probability distributions and confidence, and Boolean
116
+ requests `probabilityTrue`. These are model estimates, not native probabilities or
117
+ token logprobs. They are always tagged `model-generated`, cannot spoof provenance,
118
+ and require explicit `allowedSources: ['model-generated']` in a gate. Structured
119
+ output constrains shape; it does not calibrate those estimates.
120
+
121
+ Options also accept `generation: { maxTokens, temperature, topP, reasoningEffort }`
122
+ and `maxResponseBytes` (default 2 MiB, maximum 16 MiB), including reasoning and
123
+ arguments. Sampling/effort support remains model-specific; omission adds no sampling
124
+ defaults. Pass a raw single-attempt adapter, not a retry-wrapped registry/agent
125
+ handle. The decision runtime owns retry/deadline policy and reuses one prepared
126
+ generation across attempts. Direct bridge calls are also deadline-bounded.
127
+ For score questions in evidence mode, the wire schema requests only the
128
+ distribution and confidence. The SDK derives the public fractional score from
129
+ that validated distribution; it does not ask the model to duplicate the arithmetic.
130
+ Extra score fields and malformed distributions are rejected rather than repaired.
131
+
132
+ The SDK generation stream does not expose an authoritative response model ID or
133
+ request ID as public fields. The bridge therefore returns the requested model ID
134
+ and omits `providerRequestId`; provider-attempt telemetry remains with the wrapped
135
+ adapter. Pin model IDs when evaluation reproducibility matters. TypeSafe's native
136
+ adapter continues to preserve its actual response model ID and native evidence.
137
+
138
+ Protocol references: [OpenAI function calling](https://developers.openai.com/api/docs/guides/function-calling),
139
+ [Anthropic tool definitions](https://platform.claude.com/docs/en/agents-and-tools/tool-use/define-tools),
140
+ [Gemini Interactions](https://ai.google.dev/gemini-api/docs/interactions-overview).
141
+
142
+ ## Providers and lifecycle
143
+
144
+ Extend `DecisionAdapter`; only `evaluate()` is abstract. One invocation performs
145
+ one physical attempt. Override `resolveModel()` to declare question types and
146
+ provider-specific bounds. Mutable connection configurations must override
147
+ `prepareDecisionCall()` to bind metadata and dispatch to the same generation.
148
+ Its optional fourth argument is the invocation context: forward it during
149
+ connection/credential preparation as well as dispatch. A prepared call belongs to
150
+ one logical request; reuse that request for retries, prepare again for a new input.
151
+ Adapters must honor the request's `signal` and may report physical attempts using
152
+ the supplied SDK `ModelInvocationContext`.
153
+
154
+ `defineDecisionProviderPlugin()` declares versioned route ownership. Register
155
+ multiple plugins with `createDecisionRuntime({ providers })`, or register adapters
156
+ directly with `runtime.registerAdapter(routes, adapter)`. A route cannot have two
157
+ decision adapters; registration is atomic and returns an idempotent disposer.
158
+
159
+ This package supplies a **companion runtime**, not an `AgentRuntime` plugin. Keep
160
+ the decision runtime next to the agent runtime and call its model handles from
161
+ host workflow code or a tool. Pass `ModelInvocationContext` as the second argument
162
+ to `evaluate()` to join existing provider-attempt accounting; this runtime does
163
+ not create an independent observation exporter or fabricate attempt telemetry.
164
+
165
+ The default logical deadline is 30 seconds, including preparation, attempts and
166
+ backoff. Override with runtime `timeoutMs` or per-call `timeoutMs`. SDK retry policy
167
+ and backoff helpers are reused; the adapter's route retry policy takes precedence
168
+ over the runtime default when explicitly configured (TypeSafe inherits the runtime
169
+ policy when its `retryPolicy` is omitted). A single prepared call is reused across
170
+ retries. The runtime owns the overall deadline; the LLM bridge does not add an
171
+ independent 30-second deadline inside a runtime call. Direct LLM calls retain a
172
+ 30-second default. Transport request limits may still bound a physical attempt.
173
+ `close()` aborts active work, runs plugin cleanup, and rejects subsequent calls.
174
+ Noncooperative extension promises are detached after cancellation; their late
175
+ rejections are observed, but external side effects cannot be undone.
176
+
177
+ ### Connecting to a core agent
178
+
179
+ The current `AgentRuntime.providers` accepts generation and embedding plugins,
180
+ not decision plugins, and `AgentRuntime` has no `decisionModel()` method. A decision
181
+ task can nevertheless run inside a core tool through the public API:
182
+
183
+ ```ts
184
+ import { defineTool } from '@alvin0/ai-agent-sdk-core'
185
+
186
+ // `task` is a configured decision task with a `route` choice question.
187
+ const classifyTicket = defineTool<{ message: string }>({
188
+ name: 'classify_ticket', description: 'Classify a support ticket',
189
+ parameters: {
190
+ type: 'object', properties: { message: { type: 'string' } },
191
+ required: ['message'], additionalProperties: false,
192
+ },
193
+ parse(raw) {
194
+ if (raw === null || typeof raw !== 'object' || !('message' in raw)
195
+ || typeof raw.message !== 'string') throw new Error('message required')
196
+ return { message: raw.message }
197
+ },
198
+ async execute(args, ctx) {
199
+ const result = await task.evaluate({ message: args.message },
200
+ { signal: ctx.signal }, { ...(ctx.logger ? { logger: ctx.logger } : {}) })
201
+ return { route: result.answers.route.choice, model: result.model }
202
+ },
203
+ })
204
+ // Pass tools: [classifyTicket] when binding a core agent/session.
205
+ ```
206
+
207
+ Return the selected JSON fields to the agent, and always forward `ctx.signal`.
208
+ Closing core aborts a decision called by an active tool when that signal is
209
+ forwarded. Core does not close the independently-created decision runtime; the
210
+ host must close both owners, including decisions called outside agent runs.
211
+ Tool signals are scoped to one invocation and are aborted during cleanup, so do
212
+ not keep them for later background work.
213
+
214
+ `ToolRunContext` exposes cancellation, logging and run position; it does not
215
+ expose a complete `ModelInvocationContext` or provider-attempt accounting handle.
216
+ Forwarding its logger alone does not merge nested decision usage into the core
217
+ agent's token totals, budgets or model-call reports. A host-owned invocation
218
+ context can be passed explicitly; full core-managed admission, lifecycle,
219
+ catalogs and accounting would require a native decision capability in core.
220
+
221
+ The public integration tests cover tool output reaching the next agent step,
222
+ core-close cancellation, independent companion lifecycle, and rejection of
223
+ decision plugins in the core provider list.
224
+
225
+ Run `pnpm --filter @alvin0/ai-agent-sdk-decision-adapter test` for offline contract
226
+ tests and `test:pack` for installed-tarball verification.
227
+ For live acceptance with workspace `.env` keys, run `pnpm human:decision:all` after
228
+ building. It covers TypeSafe and OpenAI JSON Schema/tool mode, English/Vietnamese,
229
+ batches, model-generated evidence and a core agent tool. Unconfigured providers
230
+ are skipped. OpenAI defaults to `gpt-6-luna`. `OPENAI_DECISION_MODEL` (or
231
+ `OPENAI_MODEL`) may select it or a newer model; `TYPESAFE_MODEL` pins the TypeSafe
232
+ model. The runner emits safe status/usage/latency diagnostics without keys or raw
233
+ errors.
234
+
235
+ ## Reusable tasks and batches
236
+
237
+ `createDecisionTask(model, { questions, timeoutMs, context })` binds a detached,
238
+ frozen rubric to a model handle. `task.evaluate(state, { signal, timeoutMs }, context)`
239
+ overrides call options without repeating questions or connection configuration.
240
+ SDK-created immutable snapshots are reused; arbitrary frozen caller inputs still
241
+ undergo validation. Task batches share the captured rubric across queued states.
242
+ The LLM bridge and TypeSafe compile each rubric once per adapter/rubric identity,
243
+ and reuse the captured request through retries. No result cache or cross-call
244
+ credential/connection cache is introduced. Prepared calls reject requests targeting
245
+ a different provider/model before dispatch. Direct LLM prepared evaluations also
246
+ honor `timeoutMs` (30 seconds by default) and close stalled iterators without
247
+ waiting for their teardown. JSON arrays must contain only their own enumerable
248
+ indexed data properties; getters, custom iterators, symbols and auxiliary
249
+ properties are rejected without running caller hooks. LLM prompts put the fixed questions
250
+ before changing state; evidence instructions appear only in evidence mode.
251
+ Use a separate task per model/rubric combination; changing providers requires no
252
+ change to the task's question contract.
253
+
254
+ `task.evaluateBatch(states, { concurrency, signal, timeoutMs, context })` evaluates
255
+ independent states with the same rubric. `evaluateDecisionBatch(model, inputs, options)`
256
+ also accepts a different question set per input. Both preserve input order and
257
+ return `{ status: 'fulfilled', value }` or `{ status: 'rejected', reason }` per item.
258
+ They use at most four concurrent logical calls by default (configurable 1–256).
259
+ This is client-side scheduling, not a provider batch endpoint. Retries occupy the
260
+ same concurrency slot. A batch accepts at most 1,024 inputs, snapshots queued
261
+ payloads before dispatch, and does not automatically chunk or truncate data.
262
+
263
+ The batch deadline defaults to 30 seconds and includes queue time. Batch `timeoutMs`
264
+ sets that overall deadline; task `timeoutMs` or input `timeoutMs` still governs each
265
+ dispatched call, including custom handles that ignore deadlines. Item deadlines
266
+ start on dispatch, independently of the overall deadline that includes queue time.
267
+ Batch cancellation preserves caller `ModelError` reasons (including `TIMEOUT`)
268
+ and signal composition supports high concurrency without accumulating listeners
269
+ on the shared runtime/batch signal. Batch cancellation/deadline rejects the operation and stops queued
270
+ dispatches; individual item failures/cancellation remain partial results. Invalid
271
+ input/configuration rejects the batch before dispatch. Empty batches return `[]`.
272
+ For large datasets, submit application-managed chunks with explicit deadlines.
273
+ Concurrency limits active logical calls, not the memory used by queued snapshots;
274
+ choose chunk sizes based on payload size as well as item count.
275
+
276
+ ## Evidence gates
277
+
278
+ `gateChoice(answer, { minConfidence, minProbability, minMargin, allowedSources })`
279
+ requires at least one explicit threshold and accepts only when every configured
280
+ threshold passes. `minMargin` compares the selected probability to the runner-up.
281
+ `gateBoolean(answer, { falseMax, trueMin, allowedSources })` accepts false below or
282
+ at `falseMax`, true above or at `trueMin`, and abstains between them. Both return
283
+ `{ status: 'accepted', value }` or `{ status: 'abstained', reason }`.
284
+
285
+ Gates consume validated answers and default to `allowedSources: ['provider']`.
286
+ Missing required evidence or disallowed provenance causes abstention. These pure
287
+ helpers do not call a fallback model, run tools, derive confidence, or authorize
288
+ actions. Calibrate thresholds against labeled examples for each model, question
289
+ and domain. A provider confidence statistic is not an empirical accuracy estimate.
package/USE_CASES.md ADDED
@@ -0,0 +1,217 @@
1
+ # Use cases and setup
2
+
3
+ The kit separates three concerns: the provider handles transport and model
4
+ capabilities; a task binds a model to reusable questions; application code owns
5
+ thresholds, branches, ranking and actions. A decision model selects from a closed
6
+ answer space. Generated prose, open-ended tool arguments and arbitrary extraction
7
+ remain generation tasks.
8
+
9
+ ## Choosing the decision shape
10
+
11
+ | Use case | Question setup | Host workflow | Main pitfall |
12
+ | --- | --- | --- | --- |
13
+ | Intent, tool or model routing | Choice over known routes, including `other` | Gate the selected route; select a handler/model | A forced winner is not proof the request fits any route |
14
+ | Support triage / semantic features | Choice + independent booleans + ordinal scores in one request | Combine relevant answers in code | Questions do not consume each other's answers |
15
+ | Input/output/tool verification | One boolean per failure mode, with concrete true/false criteria | Block, review or continue using a three-way gate | An uncertain negative does not prove safety |
16
+ | RAG reranking / entity matching | Same relevance/match question on each query-candidate pair | Bounded batch, then filter and sort successful candidates | Choice probabilities are relative to the candidate set |
17
+ | Severity / lead prioritization | Ordered score rubric, or several independent indicators | Apply application weights and thresholds | A fractional expected score is not an integer label or a calibrated business value |
18
+ | Known-value extraction | Choice among parser-generated candidates plus `none` | Return the selected candidate's original value | A decision primitive cannot invent a missing address/date/argument |
19
+ | Hierarchical classification / cascades | Choice at each branch or a verification task after selection | Separate calls with explicit updated state and a depth/budget limit | A greedy early mistake propagates; multiplying evidence does not guarantee a calibrated path probability |
20
+
21
+ These patterns follow TypeSafe's [use-case map](https://docs.typesafe.ai/concepts/use-case-map),
22
+ [fan-out](https://docs.typesafe.ai/patterns/fan-out),
23
+ [reranking](https://docs.typesafe.ai/cookbooks/rerank_typesafe), and
24
+ [hierarchical classification](https://docs.typesafe.ai/cookbooks/hierarchical_classification)
25
+ examples. They are workflow patterns rather than provider-specific task types.
26
+
27
+ ## Setup once, reuse across consumers
28
+
29
+ For a service, create one runtime during startup, inject task handles into request
30
+ handlers/tools, and close the runtime during shutdown. For a CLI/job, use
31
+ `try/finally`. Connection configuration is separate from the rubric so consumers
32
+ do not receive credentials. Pass request-scoped invocation context explicitly
33
+ instead of storing it on a singleton task.
34
+
35
+ ```ts
36
+ import {
37
+ createDecisionRuntime, createDecisionTask, choiceQuestion, booleanQuestion,
38
+ scoreQuestion, gateChoice, gateBoolean,
39
+ } from '@alvin0/ai-agent-sdk-decision-adapter'
40
+ import { typesafePlugin } from '@alvin0/ai-agent-sdk-provider-typesafe'
41
+ import { envCredential } from '@alvin0/ai-agent-sdk-auth-node/env'
42
+
43
+ const runtime = createDecisionRuntime({
44
+ providers: [typesafePlugin({ apiKey: envCredential('TYPESAFE_API_KEY') })],
45
+ timeoutMs: 5_000,
46
+ retryPolicy: { mode: 'normal', maxRetries: 1 },
47
+ })
48
+ const model = runtime.decisionModel({ provider: 'typesafe', model: 'jev-1.13.0' })
49
+ const triage = createDecisionTask(model, {
50
+ timeoutMs: 3_000,
51
+ questions: {
52
+ route: choiceQuestion('Which team should handle this ticket?', {
53
+ billing: 'Invoices, charges and refunds',
54
+ support: 'Product failures and technical issues',
55
+ other: 'Outside the listed categories or insufficient information',
56
+ }),
57
+ severity: scoreQuestion('Impact of the technical issue, if present', [
58
+ 'Cosmetic', 'Degraded with a workaround', 'Blocked without a workaround',
59
+ ]),
60
+ refund: booleanQuestion('Is a refund explicitly requested?', {
61
+ true: 'An explicit request for money back or a credit',
62
+ false: 'No explicit refund request, including merely asking about prices',
63
+ }),
64
+ },
65
+ })
66
+ ```
67
+
68
+ Pin a tested model version for stable evaluation/calibration. An alias such as
69
+ `jev-latest` is convenient for exploration; keep `result.model` in evaluation
70
+ records so an alias change is visible. Models need no catalog allowlist. Other
71
+ providers can bind the same questions through their own decision plugins, subject
72
+ to their declared capabilities. Do not assume identical limits or probability
73
+ semantics across providers.
74
+
75
+ ### Routing and triage
76
+
77
+ ```ts
78
+ const result = await triage.evaluate({ message: 'I was charged twice; please refund.' })
79
+ const route = gateChoice(result.answers.route, {
80
+ minProbability: 0.85, minMargin: 0.2,
81
+ })
82
+ const refund = gateBoolean(result.answers.refund, { falseMax: 0.2, trueMin: 0.8 })
83
+ // Example thresholds only; measure them against your own labeled tickets.
84
+ if (route.status === 'abstained') {
85
+ // Queue for review or ask a separate model; retain the original result.
86
+ } else if (route.value === 'billing') {
87
+ // Attach refund.status/value to the billing queue, rather than issuing a refund.
88
+ } else if (route.value === 'support') {
89
+ // Read severity here; the billing path can ignore that speculative answer.
90
+ }
91
+ ```
92
+
93
+ Multiple questions in one request all see the same state. This supports multi-label
94
+ tagging through independent booleans; a single Choice returns one mutually exclusive
95
+ winner. Questions whose rubric actually depends on an earlier answer require a
96
+ second call with that answer included in the next state.
97
+
98
+ ### Guardrails and verification
99
+
100
+ Use a reusable boolean task for each input/output/tool check, passing the artifact,
101
+ relevant policy and evidence in the state. Name the question so the polarity is
102
+ clear, for example `violatesPolicy`, rather than an ambiguous `safe`.
103
+
104
+ ```ts
105
+ const verify = createDecisionTask(model, { questions: {
106
+ violatesPolicy: booleanQuestion('Does the proposed response violate the supplied policy?', {
107
+ true: 'At least one concrete policy violation is supported by the response',
108
+ false: 'The proposed response complies with every applicable supplied rule',
109
+ }),
110
+ } })
111
+ const check = await verify.evaluate({ policy: 'Do not disclose internal access tokens.', response: 'Hello!' })
112
+ const verdict = gateBoolean(check.answers.violatesPolicy, { falseMax: 0.1, trueMin: 0.8 })
113
+ // accepted false: continue; accepted true: block; abstained: review/fallback.
114
+ ```
115
+
116
+ Transport/auth failures throw; they do not become a semantic negative. The host
117
+ sets failure behavior. Decision checks supplement deterministic permission and
118
+ schema checks; the gate itself executes no action. For citations, supply both the
119
+ claim and source passage. For tool selection, validate tool arguments separately.
120
+
121
+ ### Rerank a shortlist or match entities
122
+
123
+ For a runnable example combining access/date filtering, per-facet document
124
+ selection and OpenAI answers with exact source quotes, see the
125
+ [document-selection sample](../../samples/decision-document-selection/README.md).
126
+ It also handles historical policies and abstains when evidence is incomplete.
127
+
128
+ Use one state per candidate so relevance scores do not depend on which other
129
+ candidates happen to share a request. Keep candidate IDs in the host array.
130
+
131
+ ```ts
132
+ const relevance = createDecisionTask(model, { questions: {
133
+ relevant: booleanQuestion('Does this passage directly answer the query?', {
134
+ true: 'Contains information sufficient to answer the query',
135
+ false: 'Only shares a topic or lacks the needed information',
136
+ }),
137
+ } })
138
+ const candidates = [
139
+ { id: 'a', text: 'Refunds are processed in five business days.' },
140
+ { id: 'b', text: 'Our logo is blue.' },
141
+ ]
142
+ const rows = await relevance.evaluateBatch(candidates.map(candidate => ({
143
+ query: 'When will my refund arrive?', passage: candidate.text,
144
+ })), { concurrency: 4, timeoutMs: 15_000 })
145
+ const ranked = rows.flatMap((row, index) => {
146
+ if (row.status !== 'fulfilled') return [] // Record failures separately; do not score them as zero.
147
+ const answer = row.value.answers.relevant
148
+ if (answer.probabilitySource !== 'provider' || answer.probabilityTrue === undefined) return []
149
+ return [{ candidate: candidates[index]!, score: answer.probabilityTrue, result: row.value }]
150
+ }).sort((a, b) => b.score - a.score)
151
+ ```
152
+
153
+ Retrieval produces the shortlist first; reranking cannot recover missing candidates.
154
+ Use this pattern for record linkage by supplying two candidate records and a fixed
155
+ match rubric. Prefer a labeled evaluation set to interpreting a raw probability
156
+ as a guaranteed match rate. The batch preserves order, partial failures and actual
157
+ model IDs. Configure its overall deadline for the entire queue, not just one call.
158
+
159
+ ### Staged decisions and model fallback
160
+
161
+ The host can first call a routing task, gate its evidence, then call a specialized
162
+ task with `{ originalState, selectedRoute }`. For an uncertain result, explicitly
163
+ invoke another configured decision handle or an existing generation model. Treat
164
+ semantic abstention separately from network failure. Never silently relabel an
165
+ auth/configuration failure as uncertainty. Keep each stage's result for audit.
166
+
167
+ For a taxonomy, bound depth, model calls and time; optional beam search explores
168
+ several probable branches rather than only the first winner. For candidate
169
+ extraction, let deterministic parsing produce candidates, select among their IDs,
170
+ and verify the chosen value in a second stage. Include `none`/`other` where absence
171
+ is valid. Split independent work into batches; keep dependencies sequential.
172
+
173
+ ## Provider and environment choices
174
+
175
+ | Consumer | Setup |
176
+ | --- | --- |
177
+ | Node service / agent tool | Lazy `envCredential`, shared runtime/tasks, per-request signal/context |
178
+ | Browser app | Server endpoint holding the key; inject a server-side task behind it |
179
+ | Worker / edge runtime | Inject a key from the host's secret binding or a `CredentialSource`; use Web fetch |
180
+ | Local/self-hosted API | Adapter with `baseUrl`, injected transport, explicit HTTP opt-in for local development |
181
+ | Multiple accounts/endpoints | Multiple TypeSafe plugins with distinct `id` and `routes`; bind a task to each route |
182
+ | Unit tests / offline execution | Register a deterministic `DecisionAdapter` or inject mocked fetch; keep the same task/gate code |
183
+
184
+ Credential resolution follows the underlying adapter: TypeSafe resolves per physical
185
+ attempt; prepared LLM adapters may capture credentials once per logical call.
186
+ An explicitly configured provider retry policy overrides the runtime default;
187
+ TypeSafe inherits the runtime policy when omitted. Choose one retry owner to avoid compounded attempts. Cancellation
188
+ bounds scheduling and waiting; it does not undo provider processing or host actions.
189
+ No batch cache, automatic fallback, model-specific calibration, business side effects,
190
+ or streaming dataset ingestion is implied by this package.
191
+
192
+ See TypeSafe's [confidence explanation](https://docs.typesafe.ai/confidence) for
193
+ the distribution statistic behind Choice/Score confidence. A Boolean's probability
194
+ near 0.5 is uncertain; it has no native separate confidence. The kit preserves raw
195
+ evidence and requires explicit gates rather than fabricating a shared metric.
196
+
197
+ ## Use an LLM for the same tasks
198
+
199
+ Wrap `openAiAdapter`, `anthropicAdapter`, `geminiAdapter` or another SDK model
200
+ adapter in `llmDecisionPlugin`. The [LLM setup example](./README.md#openai-anthropic-and-gemini-decisions)
201
+ registers all three. Keep the rubric and batch/workflow code; change the target
202
+ provider/model to select the implementation. Different providers can coexist with
203
+ TypeSafe in the same decision runtime.
204
+
205
+ Use JSON Schema by default, or select the tool mode for a compatible endpoint.
206
+ Generation must finish successfully before its answer is accepted; a partially
207
+ valid answer cut off by the token limit is rejected. No format repair or automatic
208
+ fallback is hidden in the adapter. Host cascades can explicitly catch failures or
209
+ handle abstention and invoke a different configured task.
210
+
211
+ LLM decisions default to values without confidence. For a routing policy that
212
+ needs raw evidence, prefer a provider that supplies it, or explicitly opt into
213
+ `evidence: 'model-generated'` and validate its behavior on your dataset. Existing
214
+ gates abstain on absent evidence and disallow these self-reported estimates by
215
+ default. Configuring structured output does not turn an LLM into a calibrated
216
+ native decision model. A batch may span an alias update; pinned IDs and per-stage
217
+ records help evaluation, while response model IDs are unavailable on this bridge.
@@ -0,0 +1,53 @@
1
+ import { MODEL_ERROR_CODES, ModelError } from "@alvin0/ai-agent-sdk-core";
2
+
3
+ //#region src/async.ts
4
+ function throwIfAborted(signal) {
5
+ if (signal.aborted) throw signal.reason instanceof ModelError ? signal.reason : new ModelError("Decision call aborted", MODEL_ERROR_CODES.ABORTED);
6
+ }
7
+ /** Settles promptly even if an extension ignores cancellation; observes its late rejection. */
8
+ function abortable(work, signal) {
9
+ return new Promise((resolve, reject) => {
10
+ const abort = () => {
11
+ cleanup();
12
+ try {
13
+ throwIfAborted(signal);
14
+ } catch (error) {
15
+ reject(error);
16
+ }
17
+ };
18
+ const cleanup = () => signal.removeEventListener("abort", abort);
19
+ work.then((value) => {
20
+ cleanup();
21
+ resolve(value);
22
+ }, (error) => {
23
+ cleanup();
24
+ reject(error);
25
+ });
26
+ signal.addEventListener("abort", abort, { once: true });
27
+ if (signal.aborted) abort();
28
+ });
29
+ }
30
+ function waitDecisionDelay(ms, signal) {
31
+ return new Promise((resolve, reject) => {
32
+ const abort = () => {
33
+ clearTimeout(timer);
34
+ cleanup();
35
+ try {
36
+ throwIfAborted(signal);
37
+ } catch (error) {
38
+ reject(error);
39
+ }
40
+ };
41
+ const cleanup = () => signal.removeEventListener("abort", abort);
42
+ const timer = setTimeout(() => {
43
+ cleanup();
44
+ resolve();
45
+ }, ms);
46
+ signal.addEventListener("abort", abort, { once: true });
47
+ if (signal.aborted) abort();
48
+ });
49
+ }
50
+
51
+ //#endregion
52
+ export { throwIfAborted as n, waitDecisionDelay as r, abortable as t };
53
+ //# sourceMappingURL=async-BPqlUFnU.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"async-BPqlUFnU.js","names":[],"sources":["../src/async.ts"],"sourcesContent":["import { ModelError, MODEL_ERROR_CODES } from '@alvin0/ai-agent-sdk-core'\n\nexport function throwIfAborted(signal: AbortSignal): void {\n if (signal.aborted) throw signal.reason instanceof ModelError ? signal.reason : new ModelError('Decision call aborted', MODEL_ERROR_CODES.ABORTED)\n}\n/** Settles promptly even if an extension ignores cancellation; observes its late rejection. */\nexport function abortable<T>(work: Promise<T>, signal: AbortSignal): Promise<T> {\n return new Promise<T>((resolve, reject) => {\n const abort = () => { cleanup(); try { throwIfAborted(signal) } catch (error) { reject(error) } }\n const cleanup = () => signal.removeEventListener('abort', abort)\n work.then(value => { cleanup(); resolve(value) }, error => { cleanup(); reject(error) })\n signal.addEventListener('abort', abort, { once: true })\n if (signal.aborted) abort()\n })\n}\nexport function waitDecisionDelay(ms: number, signal: AbortSignal): Promise<void> {\n return new Promise((resolve, reject) => {\n const abort = () => { clearTimeout(timer); cleanup(); try { throwIfAborted(signal) } catch (error) { reject(error) } }\n const cleanup = () => signal.removeEventListener('abort', abort)\n const timer = setTimeout(() => { cleanup(); resolve() }, ms)\n signal.addEventListener('abort', abort, { once: true })\n if (signal.aborted) abort()\n })\n}\n"],"mappings":";;;AAEA,SAAgB,eAAe,QAA2B;CACxD,IAAI,OAAO,SAAS,MAAM,OAAO,kBAAkB,aAAa,OAAO,SAAS,IAAI,WAAW,yBAAyB,kBAAkB,OAAO;AACnJ;;AAEA,SAAgB,UAAa,MAAkB,QAAiC;CAC9E,OAAO,IAAI,SAAY,SAAS,WAAW;EACzC,MAAM,cAAc;GAAE,QAAQ;GAAG,IAAI;IAAE,eAAe,MAAM;GAAE,SAAS,OAAO;IAAE,OAAO,KAAK;GAAE;EAAE;EAChG,MAAM,gBAAgB,OAAO,oBAAoB,SAAS,KAAK;EAC/D,KAAK,MAAK,UAAS;GAAE,QAAQ;GAAG,QAAQ,KAAK;EAAE,IAAG,UAAS;GAAE,QAAQ;GAAG,OAAO,KAAK;EAAE,CAAC;EACvF,OAAO,iBAAiB,SAAS,OAAO,EAAE,MAAM,KAAK,CAAC;EACtD,IAAI,OAAO,SAAS,MAAM;CAC5B,CAAC;AACH;AACA,SAAgB,kBAAkB,IAAY,QAAoC;CAChF,OAAO,IAAI,SAAS,SAAS,WAAW;EACtC,MAAM,cAAc;GAAE,aAAa,KAAK;GAAG,QAAQ;GAAG,IAAI;IAAE,eAAe,MAAM;GAAE,SAAS,OAAO;IAAE,OAAO,KAAK;GAAE;EAAE;EACrH,MAAM,gBAAgB,OAAO,oBAAoB,SAAS,KAAK;EAC/D,MAAM,QAAQ,iBAAiB;GAAE,QAAQ;GAAG,QAAQ;EAAE,GAAG,EAAE;EAC3D,OAAO,iBAAiB,SAAS,OAAO,EAAE,MAAM,KAAK,CAAC;EACtD,IAAI,OAAO,SAAS,MAAM;CAC5B,CAAC;AACH"}