ai-sdk-ollama 4.2.0 → 4.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +23 -0
- package/README.md +78 -13
- package/dist/functions/generate-text.d.ts.map +1 -1
- package/dist/functions/stream-text.d.ts.map +1 -1
- package/dist/index.browser.cjs +7558 -2236
- package/dist/index.browser.cjs.map +1 -1
- package/dist/index.browser.d.ts +2 -0
- package/dist/index.browser.d.ts.map +1 -1
- package/dist/index.browser.js +7558 -2236
- package/dist/index.browser.js.map +1 -1
- package/dist/index.cjs +7591 -2266
- package/dist/index.cjs.map +1 -1
- package/dist/index.d.ts +2 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +7590 -2266
- package/dist/index.js.map +1 -1
- package/dist/models/chat-language-model.d.ts +9 -2
- package/dist/models/chat-language-model.d.ts.map +1 -1
- package/dist/models/chat-language-model.test-helpers.d.ts +3 -1
- package/dist/models/chat-language-model.test-helpers.d.ts.map +1 -1
- package/dist/models/chat-result.d.ts +1 -2
- package/dist/models/chat-result.d.ts.map +1 -1
- package/dist/models/chat-stream.d.ts +2 -3
- package/dist/models/chat-stream.d.ts.map +1 -1
- package/dist/models/embedding-model.d.ts +2 -2
- package/dist/models/embedding-model.d.ts.map +1 -1
- package/dist/models/embedding-reranking-model.d.ts +3 -3
- package/dist/models/embedding-reranking-model.d.ts.map +1 -1
- package/dist/models/evaluation-model.d.ts +21 -0
- package/dist/models/evaluation-model.d.ts.map +1 -0
- package/dist/ollama-client.d.ts +42 -0
- package/dist/ollama-client.d.ts.map +1 -0
- package/dist/provider.browser.d.ts.map +1 -1
- package/dist/provider.d.ts +11 -10
- package/dist/provider.d.ts.map +1 -1
- package/dist/tool/web-fetch.d.ts +6 -4
- package/dist/tool/web-fetch.d.ts.map +1 -1
- package/dist/tool/web-search.d.ts +8 -6
- package/dist/tool/web-search.d.ts.map +1 -1
- package/dist/utils/json-schema-coercion.d.ts +3 -4
- package/dist/utils/json-schema-coercion.d.ts.map +1 -1
- package/dist/utils/json-text-repair.d.ts +9 -6
- package/dist/utils/json-text-repair.d.ts.map +1 -1
- package/dist/utils/last-user-text.d.ts +7 -0
- package/dist/utils/last-user-text.d.ts.map +1 -0
- package/dist/utils/object-generation-reliability.d.ts +3 -4
- package/dist/utils/object-generation-reliability.d.ts.map +1 -1
- package/dist/utils/ollama-error.d.ts +2 -0
- package/dist/utils/ollama-error.d.ts.map +1 -1
- package/package.json +18 -21
package/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,28 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 4.4.0
|
|
4
|
+
|
|
5
|
+
### Minor Changes
|
|
6
|
+
|
|
7
|
+
- 004f6b4: Add `ollama.evaluationModel()` for Ollama decision models such as `nimble` and `tev1`. Use it with the AI SDK's `experimental_evaluate` to get typed choice, boolean and score answers with probabilities from one local call.
|
|
8
|
+
|
|
9
|
+
- `OllamaClient` accepts an optional `systemone()` method.
|
|
10
|
+
- `generateText` and `streamText` restate the latest user message, text parts included, when they synthesise a reply from tool results.
|
|
11
|
+
- JSON repair serialises object values for string fields as JSON.
|
|
12
|
+
- Supports `ai` 7.0.103+ and `ollama` 0.6.4.
|
|
13
|
+
|
|
14
|
+
## 4.3.0
|
|
15
|
+
|
|
16
|
+
### Minor Changes
|
|
17
|
+
|
|
18
|
+
- 371f515: Track AI SDK 7.0.95 and open the provider up to custom Ollama clients.
|
|
19
|
+
|
|
20
|
+
- The `ai` peer range moves to `^7.0.95`, with `@ai-sdk/provider` and `@ai-sdk/provider-utils` updated to match.
|
|
21
|
+
- `OllamaClient` is a structural contract, so the official client or a custom adapter can be injected in both Node.js and browser builds. Streaming methods use the newly exported `AbortableStream` type.
|
|
22
|
+
- Per-request `AbortSignal`s are forwarded to the client, and an aborted request settles as a rejection across the reliability retry and fallback paths.
|
|
23
|
+
- Returned `message.thinking` is surfaced as AI SDK reasoning independently of the request's `think` setting, including on terminal stream chunks, and streamed text and reasoning blocks close symmetrically.
|
|
24
|
+
- Output usage carries the total token count Ollama reports and leaves the text/reasoning split unset, matching what the API provides.
|
|
25
|
+
|
|
3
26
|
## 4.2.0
|
|
4
27
|
|
|
5
28
|
### Minor Changes
|
package/README.md
CHANGED
|
@@ -154,6 +154,7 @@ export OLLAMA_API_KEY="your_api_key_here"
|
|
|
154
154
|
- [Combining Tools with Structured Output](#combining-tools-with-structured-output)
|
|
155
155
|
- [Simple and Predictable](#simple-and-predictable)
|
|
156
156
|
- [Reranking](#reranking)
|
|
157
|
+
- [Decision Models](#decision-models)
|
|
157
158
|
- [Streaming Utilities](#streaming-utilities)
|
|
158
159
|
- [Smooth Stream](#smooth-stream)
|
|
159
160
|
- [Partial JSON Parsing](#partial-json-parsing)
|
|
@@ -481,6 +482,46 @@ ranking.forEach((item, i) => {
|
|
|
481
482
|
|
|
482
483
|
**Recommended Models**: `embeddinggemma` (best score separation), `nomic-embed-text`, `bge-m3`
|
|
483
484
|
|
|
485
|
+
## Decision Models
|
|
486
|
+
|
|
487
|
+
Ollama 0.35 serves decision models (`nimble`, `tev1`, `tev1:0.8b`) on `/v1/systemone`. You ask typed questions about some state and get each answer with its probabilities in one local call. Use them for triage, routing and moderation through the AI SDK's `experimental_evaluate`:
|
|
488
|
+
|
|
489
|
+
```typescript
|
|
490
|
+
import { experimental_evaluate as evaluate } from 'ai';
|
|
491
|
+
import { ollama } from 'ai-sdk-ollama';
|
|
492
|
+
|
|
493
|
+
const { answers } = await evaluate({
|
|
494
|
+
model: ollama.evaluationModel('nimble'),
|
|
495
|
+
state: { ticket: 'I was charged twice. Please refund the extra payment.' },
|
|
496
|
+
questions: {
|
|
497
|
+
team: {
|
|
498
|
+
type: 'choice',
|
|
499
|
+
instructions: 'Which team should handle this ticket?',
|
|
500
|
+
criteria: {
|
|
501
|
+
billing: 'Payments and refunds',
|
|
502
|
+
technical: null,
|
|
503
|
+
other: null,
|
|
504
|
+
},
|
|
505
|
+
},
|
|
506
|
+
refund: {
|
|
507
|
+
type: 'boolean',
|
|
508
|
+
instructions: 'Does the customer ask for a refund?',
|
|
509
|
+
},
|
|
510
|
+
urgency: {
|
|
511
|
+
type: 'score',
|
|
512
|
+
instructions: 'How urgent is this ticket?',
|
|
513
|
+
criteria: ['Routine', 'Soon', 'Urgent'],
|
|
514
|
+
},
|
|
515
|
+
},
|
|
516
|
+
});
|
|
517
|
+
|
|
518
|
+
answers.team.choice; // 'billing'
|
|
519
|
+
answers.refund.probability; // P(true), e.g. 0.99
|
|
520
|
+
answers.urgency.score; // 0 to 2, e.g. 0.8
|
|
521
|
+
```
|
|
522
|
+
|
|
523
|
+
Ollama names boolean questions `noul`, and the provider maps them for you. Ollama's per-question `confidence` lands in `providerMetadata.ollama.confidence`. Pull a model first with `ollama pull nimble`.
|
|
524
|
+
|
|
484
525
|
## Streaming Utilities
|
|
485
526
|
|
|
486
527
|
### Smooth Stream
|
|
@@ -740,7 +781,9 @@ const ollama = createOllama({ apiKey: process.env.OLLAMA_API_KEY });
|
|
|
740
781
|
|
|
741
782
|
### Using Existing Ollama Client
|
|
742
783
|
|
|
743
|
-
You can also pass an existing Ollama client instance to reuse your configuration
|
|
784
|
+
You can also pass an existing Ollama client instance to reuse your configuration.
|
|
785
|
+
The provider accepts the exported structural `OllamaClient` interface, so you
|
|
786
|
+
can inject the official client or a custom adapter in Node.js and browser builds.
|
|
744
787
|
|
|
745
788
|
```typescript
|
|
746
789
|
import { Ollama } from 'ollama';
|
|
@@ -756,13 +799,36 @@ const existingClient = new Ollama({
|
|
|
756
799
|
const ollamaSdk = createOllama({ client: existingClient });
|
|
757
800
|
|
|
758
801
|
// Use both clients as needed
|
|
759
|
-
await
|
|
802
|
+
await existingClient.list(); // Direct Ollama operations
|
|
760
803
|
const { text } = await generateText({
|
|
761
804
|
model: ollamaSdk('llama3.2'),
|
|
762
805
|
prompt: 'Hello!',
|
|
763
806
|
});
|
|
764
807
|
```
|
|
765
808
|
|
|
809
|
+
#### Custom client adapters
|
|
810
|
+
|
|
811
|
+
`OllamaClient` is structural, so any object with the right shape works — no
|
|
812
|
+
`Ollama` instance required. Streaming methods return an `AbortableStream`, an
|
|
813
|
+
async iterable with an `abort()`:
|
|
814
|
+
|
|
815
|
+
```typescript
|
|
816
|
+
import type { AbortableStream, OllamaClient } from 'ai-sdk-ollama';
|
|
817
|
+
import { createOllama } from 'ai-sdk-ollama';
|
|
818
|
+
|
|
819
|
+
const client: OllamaClient = {
|
|
820
|
+
chat: myChat, // returns a ChatResponse, or an AbortableStream of them
|
|
821
|
+
embed: myEmbed,
|
|
822
|
+
webSearch: myWebSearch,
|
|
823
|
+
webFetch: myWebFetch,
|
|
824
|
+
};
|
|
825
|
+
|
|
826
|
+
const ollama = createOllama({ client });
|
|
827
|
+
```
|
|
828
|
+
|
|
829
|
+
Per-request cancellation is forwarded to the client as an `AbortSignal`, and an
|
|
830
|
+
aborted request rejects rather than resolving with a retried or fallback result.
|
|
831
|
+
|
|
766
832
|
### Structured Output
|
|
767
833
|
|
|
768
834
|
```typescript
|
|
@@ -918,39 +984,38 @@ const { output: custom } = await generateText({
|
|
|
918
984
|
|
|
919
985
|
### Reasoning Support
|
|
920
986
|
|
|
921
|
-
|
|
987
|
+
Use `think` to request reasoning generation from models that support it. Returned `message.thinking` is exposed as AI SDK reasoning, separately from final text, in both generation results and streams (including terminal chunks).
|
|
922
988
|
|
|
923
989
|
```typescript
|
|
924
990
|
import { ollama } from 'ai-sdk-ollama';
|
|
925
991
|
import { generateText } from 'ai';
|
|
926
992
|
|
|
927
993
|
// Enable reasoning for models that support it (e.g., deepseek-r1:7b)
|
|
928
|
-
const model = ollama('deepseek-r1:7b', {
|
|
994
|
+
const model = ollama('deepseek-r1:7b', { think: true });
|
|
929
995
|
|
|
930
996
|
// Generate text with reasoning
|
|
931
|
-
const { text } = await generateText({
|
|
997
|
+
const { text, reasoningText } = await generateText({
|
|
932
998
|
model,
|
|
933
999
|
prompt:
|
|
934
1000
|
'Solve: If I have 3 boxes, each with 4 smaller boxes, and each smaller box has 5 items, how many items total?',
|
|
935
1001
|
});
|
|
936
1002
|
|
|
937
1003
|
console.log('Answer:', text);
|
|
938
|
-
|
|
939
|
-
// <think>
|
|
940
|
-
// First, I'll calculate the number of smaller boxes: 3 × 4 = 12
|
|
941
|
-
// Then, the total items: 12 × 5 = 60
|
|
942
|
-
// </think>
|
|
943
|
-
// You have 60 items in total.
|
|
1004
|
+
console.log('Reasoning:', reasoningText);
|
|
944
1005
|
|
|
945
1006
|
// Compare with reasoning disabled
|
|
946
|
-
const modelNoReasoning = ollama('deepseek-r1:7b', {
|
|
1007
|
+
const modelNoReasoning = ollama('deepseek-r1:7b', { think: false });
|
|
947
1008
|
const { text: noReasoningText } = await generateText({
|
|
948
1009
|
model: modelNoReasoning,
|
|
949
1010
|
prompt: 'Calculate 3 × 4 × 5',
|
|
950
1011
|
});
|
|
951
|
-
|
|
1012
|
+
console.log('Answer:', noReasoningText);
|
|
952
1013
|
```
|
|
953
1014
|
|
|
1015
|
+
`think` controls the request, not filtering of the response. If the server returns thinking with `think` omitted or `false`, the adapter still preserves it as reasoning; it does not turn it into final text. If no thinking is returned, no reasoning content is added.
|
|
1016
|
+
|
|
1017
|
+
Ollama supplies a combined output token count. The adapter preserves that total and the raw counters, but leaves the text/reasoning token breakdown unknown instead of attributing all output tokens to text.
|
|
1018
|
+
|
|
954
1019
|
**Recommended Reasoning Models**:
|
|
955
1020
|
|
|
956
1021
|
- `deepseek-r1:7b` - Balanced performance and reasoning capability (5GB)
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"generate-text.d.ts","sourceRoot":"","sources":["../../src/functions/generate-text.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AAEH,OAAO,EAAE,YAAY,IAAI,aAAa,EAAe,MAAM,IAAI,CAAC;
|
|
1
|
+
{"version":3,"file":"generate-text.d.ts","sourceRoot":"","sources":["../../src/functions/generate-text.ts"],"names":[],"mappings":"AAAA;;;;;GAKG;AAEH,OAAO,EAAE,YAAY,IAAI,aAAa,EAAe,MAAM,IAAI,CAAC;AAsBhE;;GAEG;AACH,MAAM,WAAW,eAAe;IAC9B;;;OAGG;IACH,eAAe,CAAC,EAAE,OAAO,CAAC;IAE1B;;OAEG;IACH,eAAe,CAAC,EAAE,MAAM,CAAC;IAEzB;;;OAGG;IACH,oBAAoB,CAAC,EAAE,MAAM,CAAC;IAE9B;;;OAGG;IACH,iBAAiB,CAAC,EAAE,MAAM,CAAC;IAE3B;;;;;;;;;;;OAWG;IACH,+BAA+B,CAAC,EAAE,OAAO,CAAC;CAC3C;AAED;;GAEG;AACH,MAAM,MAAM,mBAAmB,GAAG,UAAU,CAAC,OAAO,aAAa,CAAC,CAAC,CAAC,CAAC,GAAG;IACtE;;OAEG;IACH,eAAe,CAAC,EAAE,eAAe,CAAC;CACnC,CAAC;AAEF;;;;;;;GAOG;AACH,wBAAsB,YAAY,CAChC,OAAO,EAAE,mBAAmB,GAC3B,OAAO,CAAC,OAAO,CAAC,UAAU,CAAC,OAAO,aAAa,CAAC,CAAC,CAAC,CAyNpD"}
|
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"file":"stream-text.d.ts","sourceRoot":"","sources":["../../src/functions/stream-text.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,UAAU,IAAI,WAAW,EAAe,MAAM,IAAI,CAAC;
|
|
1
|
+
{"version":3,"file":"stream-text.d.ts","sourceRoot":"","sources":["../../src/functions/stream-text.ts"],"names":[],"mappings":"AAAA,OAAO,EAAE,UAAU,IAAI,WAAW,EAAe,MAAM,IAAI,CAAC;AAG5D,KAAK,mBAAmB,GAAG,UAAU,CAAC,OAAO,WAAW,CAAC,CAAC,CAAC,CAAC,CAAC;AAsC7D,MAAM,MAAM,iBAAiB,GAAG,mBAAmB,GAAG;IACpD,eAAe,CAAC,EAAE;QAChB;;WAEG;QACH,iBAAiB,CAAC,EAAE,OAAO,CAAC;QAC5B;;WAEG;QACH,wBAAwB,CAAC,EAAE,OAAO,CAAC;QACnC;;WAEG;QACH,eAAe,CAAC,EAAE,MAAM,CAAC;QACzB;;WAEG;QACH,gBAAgB,CAAC,EAAE,MAAM,CAAC;KAC3B,CAAC;CACH,CAAC;AAUF,wBAAsB,UAAU,CAC9B,OAAO,EAAE,iBAAiB,GACzB,OAAO,CAAC,OAAO,CAAC,UAAU,CAAC,OAAO,WAAW,CAAC,CAAC,CAAC,CAkclD"}
|